{
  "name": "MargIQ Workflow-Aware LLM Routing Benchmark",
  "url": "https://getmargiq.com/benchmarks/workflow-aware-llm-routing",
  "benchmarkType": "controlled_sandbox",
  "publisher": {
    "name": "MargIQ",
    "url": "https://getmargiq.com/"
  },
  "author": {
    "name": "Rakshith Hegde",
    "role": "Founder, MargIQ"
  },
  "datePublished": "2026-07-12",
  "dateModified": "2026-07-27",
  "disclosure": "This was a MargIQ-controlled sandbox benchmark, not customer production traffic.",
  "summary": "MargIQ routed a routine GPT-5 order lookup to GPT-4.1 mini at 74.83% lower measured model cost while retaining GPT-5 for a safety-sensitive refund request in the same workflow.",
  "methodology": {
    "traffic": "Controlled OpenRouter traffic through an OpenAI-compatible client wrapped by MargIQ.",
    "requestedModel": "openai/gpt-5",
    "decisionBasis": [
      "recurring workflow evidence",
      "model cost",
      "output quality",
      "latency",
      "request risk"
    ],
    "measurementScope": "Path-level transaction evidence from the retained benchmark snapshot."
  },
  "coverage": {
    "workflowFamiliesEvaluated": 15,
    "safetyAndGeneralizationScenarios": 21,
    "retainedSnapshotRequests": 58,
    "activeDays": 1,
    "workflowCategories": [
      "classification",
      "extraction",
      "summarization",
      "writing",
      "customer chat",
      "tool use",
      "privacy",
      "security-sensitive work"
    ]
  },
  "verifiedRoutes": [
    {
      "workflow": "Ecommerce order-support triage",
      "path": "Routine order-status lookup",
      "requestedModel": "openai/gpt-5",
      "selectedModel": "openai/gpt-4.1-mini",
      "requestedModelBaselineUsd": 0.000441,
      "selectedModelCostUsd": 0.000111,
      "measuredSavingsUsd": 0.00033,
      "measuredModelCostReductionPercent": 74.83,
      "result": "Lower-cost route activated"
    },
    {
      "workflow": "Customer-message intent routing",
      "path": "Strict low-risk intent classification",
      "requestedModel": "openai/gpt-5",
      "selectedModel": "openai/gpt-4o-mini",
      "requestedModelBaselineUsd": 0.000259,
      "selectedModelCostUsd": 0.000024,
      "measuredSavingsUsd": 0.000235,
      "measuredModelCostReductionPercent": 90.73,
      "result": "Lower-cost route activated"
    }
  ],
  "qualityProtection": [
    {
      "workflow": "Ecommerce order-support triage",
      "path": "Safety-sensitive refund and escalation",
      "requestedModel": "openai/gpt-5",
      "selectedModel": "openai/gpt-5",
      "measuredSavingsUsd": 0,
      "result": "Requested model retained"
    },
    {
      "workflow": "Support-ticket classification",
      "result": "Optimization blocked because decision-bearing fields lacked an authoritative quality definition"
    }
  ],
  "limitations": [
    "Controlled sandbox benchmark, not customer production traffic.",
    "One active day and two recorded lower-cost decisions in the retained snapshot.",
    "Path-level reductions do not represent account-wide or guaranteed savings.",
    "Results depend on traffic mix, token usage, available models, and provider pricing.",
    "Latency observations were directional and not a controlled performance experiment.",
    "Long-term drift, failure rates, and human-review impact were not measured."
  ],
  "citation": "Hegde, Rakshith. MargIQ Workflow-Aware LLM Routing Benchmark. MargIQ, 12 July 2026. Updated 27 July 2026. https://getmargiq.com/benchmarks/workflow-aware-llm-routing"
}
