{
  "scope": "Exploratory replay of the same frozen 500 routing cases and all 30 APIs. No new API requests. The validation-selected threshold predates the test. Every case was actually sent to every API in the source experiment. Hybrid call avoidance, cost and latency are replay estimates, not live savings. Not an audit-workflow benchmark or evidence of equal production quality.",
  "source_summary_sha256": "09a227fe0eeb6c710bc83e101c33213b593cd493c08c219df7fc071250c0384f",
  "source_evidence_manifest_sha256": "18f74fedc18d6e7b43c175644df8833e9ceb31a3e696d73f071cc04401d8095e",
  "analysis_tool_sha256": "7fdb790db53de33fc23c9a2aa95f8db7d6940f4eced928c12a6446f19bfb7097",
  "threshold_selection": {
    "threshold": 0.8502035140991211,
    "accepted": 193,
    "correct": 192,
    "validation_n": 200
  },
  "cost_assumptions": {
    "allocated_hourly_usd": 0.5,
    "utilization": 1
  },
  "latency_method": "For accepted cases use recorded warm local duration; for escalations add the local duration to the recorded full API duration. This does not model queueing, load-dependent provider behavior, or cold starts.",
  "quality_method": "Keep every failure in the denominator. Report paired gains and losses and separate 95% Wilson accuracy intervals; these are descriptive comparisons, not proof of non-inferiority. The original 1% accepted-error validation target was missed on test (8/490 = 1.63%).",
  "entries": [
    {
      "model": "openai/gpt-4.1-nano",
      "name": "OpenAI: GPT-4.1 Nano",
      "direct": {
        "n": 500,
        "correct": 494,
        "accuracy": 0.988,
        "accuracy_wilson_95": [
          0.9740696388835527,
          0.9944890048259724
        ],
        "total_seconds": 330.5819375384017,
        "median_seconds": 0.6051433540415019,
        "p95_seconds": 1.0113971249666065
      },
      "hybrid_replay": {
        "n": 500,
        "correct": 491,
        "accuracy": 0.982,
        "accuracy_wilson_95": [
          0.9661483290912607,
          0.990501806703803
        ],
        "total_seconds": 89.89647071476793,
        "median_seconds": 0.1622084794798866,
        "p95_seconds": 0.22851016698405147
      },
      "local_accepted": 490,
      "local_accepted_errors": 8,
      "escalations": 10,
      "avoided_api_call_fraction": 0.98,
      "direct_cost_per_1000": 0.014042,
      "hybrid_cost_per_1000": 0.023355766078266792,
      "fallback_charge_usd": 0.0001428,
      "cost_savings_fraction": -0.6632791680862264,
      "accuracy_change_pp": -0.6000000000000005,
      "paired_correctness": {
        "gained": 4,
        "lost": 7,
        "unchanged_correctness": 489
      },
      "paid_tokens_in_replay": {
        "prompt_tokens": 1344,
        "completion_tokens": 21
      }
    },
    {
      "model": "openai/gpt-4.1-mini",
      "name": "OpenAI: GPT-4.1 Mini",
      "direct": {
        "n": 500,
        "correct": 496,
        "accuracy": 0.992,
        "accuracy_wilson_95": [
          0.9796129641658948,
          0.9968846848199377
        ],
        "total_seconds": 360.4603449405404,
        "median_seconds": 0.6674674169917125,
        "p95_seconds": 1.064537999976892
      },
      "hybrid_replay": {
        "n": 500,
        "correct": 492,
        "accuracy": 0.984,
        "accuracy_wilson_95": [
          0.9687488914666058,
          0.9918707469666117
        ],
        "total_seconds": 91.48217375576496,
        "median_seconds": 0.1622084794798866,
        "p95_seconds": 0.22851016698405147
      },
      "local_accepted": 490,
      "local_accepted_errors": 8,
      "escalations": 10,
      "avoided_api_call_fraction": 0.98,
      "direct_cost_per_1000": 0.056152,
      "hybrid_cost_per_1000": 0.024209366078266792,
      "fallback_charge_usd": 0.0005696,
      "cost_savings_fraction": 0.56886012825426,
      "accuracy_change_pp": -0.8000000000000007,
      "paired_correctness": {
        "gained": 3,
        "lost": 7,
        "unchanged_correctness": 490
      },
      "paid_tokens_in_replay": {
        "prompt_tokens": 1344,
        "completion_tokens": 20
      }
    },
    {
      "model": "openai/gpt-4.1",
      "name": "OpenAI: GPT-4.1",
      "direct": {
        "n": 500,
        "correct": 495,
        "accuracy": 0.99,
        "accuracy_wilson_95": [
          0.9768069002442693,
          0.9957212461034096
        ],
        "total_seconds": 359.2503143767826,
        "median_seconds": 0.6558008959982544,
        "p95_seconds": 1.0276261669932865
      },
      "hybrid_replay": {
        "n": 500,
        "correct": 492,
        "accuracy": 0.984,
        "accuracy_wilson_95": [
          0.9687488914666058,
          0.9918707469666117
        ],
        "total_seconds": 90.32007071585394,
        "median_seconds": 0.1622084794798866,
        "p95_seconds": 0.22851016698405147
      },
      "local_accepted": 490,
      "local_accepted_errors": 8,
      "escalations": 10,
      "avoided_api_call_fraction": 0.98,
      "direct_cost_per_1000": 0.28076,
      "hybrid_cost_per_1000": 0.02876616607826679,
      "fallback_charge_usd": 0.002848,
      "cost_savings_fraction": 0.8975417934240391,
      "accuracy_change_pp": -0.6000000000000005,
      "paired_correctness": {
        "gained": 4,
        "lost": 7,
        "unchanged_correctness": 489
      },
      "paid_tokens_in_replay": {
        "prompt_tokens": 1344,
        "completion_tokens": 20
      }
    },
    {
      "model": "openai/gpt-4o-mini",
      "name": "OpenAI: GPT-4o-mini",
      "direct": {
        "n": 500,
        "correct": 492,
        "accuracy": 0.984,
        "accuracy_wilson_95": [
          0.9687488914666058,
          0.9918707469666117
        ],
        "total_seconds": 263.8602219266468,
        "median_seconds": 0.5019848960218951,
        "p95_seconds": 0.6848381250165403
      },
      "hybrid_replay": {
        "n": 500,
        "correct": 492,
        "accuracy": 0.984,
        "accuracy_wilson_95": [
          0.9687488914666058,
          0.9918707469666117
        ],
        "total_seconds": 88.25850067386637,
        "median_seconds": 0.1622084794798866,
        "p95_seconds": 0.22851016698405147
      },
      "local_accepted": 490,
      "local_accepted_errors": 8,
      "escalations": 10,
      "avoided_api_call_fraction": 0.98,
      "direct_cost_per_1000": 0.0204822,
      "hybrid_cost_per_1000": 0.023485366078266793,
      "fallback_charge_usd": 0.0002076,
      "cost_savings_fraction": -0.1466232181243614,
      "accuracy_change_pp": 0.0,
      "paired_correctness": {
        "gained": 6,
        "lost": 6,
        "unchanged_correctness": 488
      },
      "paid_tokens_in_replay": {
        "prompt_tokens": 1344,
        "completion_tokens": 10
      }
    },
    {
      "model": "openai/gpt-5.6-luna",
      "name": "OpenAI: GPT-5.6 Luna",
      "direct": {
        "n": 500,
        "correct": 491,
        "accuracy": 0.982,
        "accuracy_wilson_95": [
          0.9661483290912607,
          0.990501806703803
        ],
        "total_seconds": 858.9793787939707,
        "median_seconds": 1.1447858744941186,
        "p95_seconds": 3.4226961249951273
      },
      "hybrid_replay": {
        "n": 500,
        "correct": 492,
        "accuracy": 0.984,
        "accuracy_wilson_95": [
          0.9687488914666058,
          0.9918707469666117
        ],
        "total_seconds": 99.94951013365062,
        "median_seconds": 0.1622084794798866,
        "p95_seconds": 0.22851016698405147
      },
      "local_accepted": 490,
      "local_accepted_errors": 8,
      "escalations": 10,
      "avoided_api_call_fraction": 0.98,
      "direct_cost_per_1000": 0.032276,
      "hybrid_cost_per_1000": 0.023723766078266793,
      "fallback_charge_usd": 0.0003268,
      "cost_savings_fraction": 0.26497192718221607,
      "accuracy_change_pp": 0.20000000000000018,
      "paired_correctness": {
        "gained": 8,
        "lost": 7,
        "unchanged_correctness": 485
      },
      "paid_tokens_in_replay": {
        "prompt_tokens": 1334,
        "completion_tokens": 50
      }
    },
    {
      "model": "openai/gpt-5.6-sol",
      "name": "OpenAI: GPT-5.6 Sol",
      "direct": {
        "n": 500,
        "correct": 495,
        "accuracy": 0.99,
        "accuracy_wilson_95": [
          0.9768069002442693,
          0.9957212461034096
        ],
        "total_seconds": 1662.7701584976166,
        "median_seconds": 3.400876541476464,
        "p95_seconds": 6.904222582990769
      },
      "hybrid_replay": {
        "n": 500,
        "correct": 491,
        "accuracy": 0.982,
        "accuracy_wilson_95": [
          0.9661483290912607,
          0.990501806703803
        ],
        "total_seconds": 119.4184785077232,
        "median_seconds": 0.1622084794798866,
        "p95_seconds": 0.22851016698405147
      },
      "local_accepted": 490,
      "local_accepted_errors": 8,
      "escalations": 10,
      "avoided_api_call_fraction": 0.98,
      "direct_cost_per_1000": 0.31276,
      "hybrid_cost_per_1000": 0.029406166078266793,
      "fallback_charge_usd": 0.003168,
      "cost_savings_fraction": 0.9059784944421704,
      "accuracy_change_pp": -0.8000000000000007,
      "paired_correctness": {
        "gained": 3,
        "lost": 7,
        "unchanged_correctness": 490
      },
      "paid_tokens_in_replay": {
        "prompt_tokens": 1334,
        "completion_tokens": 50
      }
    },
    {
      "model": "anthropic/claude-haiku-4.5",
      "name": "Anthropic: Claude Haiku 4.5",
      "direct": {
        "n": 500,
        "correct": 494,
        "accuracy": 0.988,
        "accuracy_wilson_95": [
          0.9740696388835527,
          0.9944890048259724
        ],
        "total_seconds": 409.28693951509194,
        "median_seconds": 0.7780484375252854,
        "p95_seconds": 0.9811019590124488
      },
      "hybrid_replay": {
        "n": 500,
        "correct": 492,
        "accuracy": 0.984,
        "accuracy_wilson_95": [
          0.9687488914666058,
          0.9918707469666117
        ],
        "total_seconds": 91.14177325787023,
        "median_seconds": 0.1622084794798866,
        "p95_seconds": 0.22851016698405147
      },
      "local_accepted": 490,
      "local_accepted_errors": 8,
      "escalations": 10,
      "avoided_api_call_fraction": 0.98,
      "direct_cost_per_1000": 0.160436,
      "hybrid_cost_per_1000": 0.02632016607826679,
      "fallback_charge_usd": 0.001625,
      "cost_savings_fraction": 0.8359460091359371,
      "accuracy_change_pp": -0.40000000000000036,
      "paired_correctness": {
        "gained": 5,
        "lost": 7,
        "unchanged_correctness": 488
      },
      "paid_tokens_in_replay": {
        "prompt_tokens": 1375,
        "completion_tokens": 50
      }
    },
    {
      "model": "anthropic/claude-sonnet-5",
      "name": "Anthropic: Claude Sonnet 5",
      "direct": {
        "n": 500,
        "correct": 493,
        "accuracy": 0.986,
        "accuracy_wilson_95": [
          0.9713869308622147,
          0.9932022102091564
        ],
        "total_seconds": 887.6002512009,
        "median_seconds": 1.718754583504051,
        "p95_seconds": 2.1111920829862356
      },
      "hybrid_replay": {
        "n": 500,
        "correct": 492,
        "accuracy": 0.984,
        "accuracy_wilson_95": [
          0.9687488914666058,
          0.9918707469666117
        ],
        "total_seconds": 101.60435513081029,
        "median_seconds": 0.1622084794798866,
        "p95_seconds": 0.22851016698405147
      },
      "local_accepted": 490,
      "local_accepted_errors": 8,
      "escalations": 10,
      "avoided_api_call_fraction": 0.98,
      "direct_cost_per_1000": 0.38084,
      "hybrid_cost_per_1000": 0.03079416607826679,
      "fallback_charge_usd": 0.003862,
      "cost_savings_fraction": 0.9191414607754784,
      "accuracy_change_pp": -0.20000000000000018,
      "paired_correctness": {
        "gained": 6,
        "lost": 7,
        "unchanged_correctness": 487
      },
      "paid_tokens_in_replay": {
        "prompt_tokens": 1781,
        "completion_tokens": 30
      }
    },
    {
      "model": "google/gemini-2.5-flash-lite",
      "name": "Google: Gemini 2.5 Flash Lite",
      "direct": {
        "n": 500,
        "correct": 494,
        "accuracy": 0.988,
        "accuracy_wilson_95": [
          0.9740696388835527,
          0.9944890048259724
        ],
        "total_seconds": 224.3929377865279,
        "median_seconds": 0.42386018752586097,
        "p95_seconds": 0.5583864590153098
      },
      "hybrid_replay": {
        "n": 500,
        "correct": 492,
        "accuracy": 0.984,
        "accuracy_wilson_95": [
          0.9687488914666058,
          0.9918707469666117
        ],
        "total_seconds": 87.21639413083903,
        "median_seconds": 0.1622084794798866,
        "p95_seconds": 0.22851016698405147
      },
      "local_accepted": 490,
      "local_accepted_errors": 8,
      "escalations": 10,
      "avoided_api_call_fraction": 0.98,
      "direct_cost_per_1000": 0.0130096,
      "hybrid_cost_per_1000": 0.023334566078266793,
      "fallback_charge_usd": 0.0001322,
      "cost_savings_fraction": -0.7936420857110744,
      "accuracy_change_pp": -0.40000000000000036,
      "paired_correctness": {
        "gained": 4,
        "lost": 6,
        "unchanged_correctness": 490
      },
      "paid_tokens_in_replay": {
        "prompt_tokens": 1274,
        "completion_tokens": 12
      }
    },
    {
      "model": "google/gemini-2.5-flash",
      "name": "Google: Gemini 2.5 Flash",
      "direct": {
        "n": 500,
        "correct": 493,
        "accuracy": 0.986,
        "accuracy_wilson_95": [
          0.9713869308622147,
          0.9932022102091564
        ],
        "total_seconds": 248.76296508917585,
        "median_seconds": 0.46842281249701045,
        "p95_seconds": 0.6624899580492638
      },
      "hybrid_replay": {
        "n": 500,
        "correct": 491,
        "accuracy": 0.982,
        "accuracy_wilson_95": [
          0.9661483290912607,
          0.990501806703803
        ],
        "total_seconds": 90.00186787982238,
        "median_seconds": 0.1622084794798866,
        "p95_seconds": 0.22851016698405147
      },
      "local_accepted": 490,
      "local_accepted_errors": 8,
      "escalations": 10,
      "avoided_api_call_fraction": 0.98,
      "direct_cost_per_1000": 0.0404688,
      "hybrid_cost_per_1000": 0.02390456607826679,
      "fallback_charge_usd": 0.0004172,
      "cost_savings_fraction": 0.4093087494991996,
      "accuracy_change_pp": -0.40000000000000036,
      "paired_correctness": {
        "gained": 5,
        "lost": 7,
        "unchanged_correctness": 488
      },
      "paid_tokens_in_replay": {
        "prompt_tokens": 1274,
        "completion_tokens": 14
      }
    },
    {
      "model": "google/gemini-3.1-flash-lite",
      "name": "Google: Gemini 3.1 Flash Lite",
      "direct": {
        "n": 500,
        "correct": 496,
        "accuracy": 0.992,
        "accuracy_wilson_95": [
          0.9796129641658948,
          0.9968846848199377
        ],
        "total_seconds": 341.9583707738784,
        "median_seconds": 0.6557884794892743,
        "p95_seconds": 0.9215647499659099
      },
      "hybrid_replay": {
        "n": 500,
        "correct": 492,
        "accuracy": 0.984,
        "accuracy_wilson_95": [
          0.9687488914666058,
          0.9918707469666117
        ],
        "total_seconds": 90.17406050575664,
        "median_seconds": 0.1622084794798866,
        "p95_seconds": 0.22851016698405147
      },
      "local_accepted": 490,
      "local_accepted_errors": 8,
      "escalations": 10,
      "avoided_api_call_fraction": 0.98,
      "direct_cost_per_1000": 0.032774,
      "hybrid_cost_per_1000": 0.023737166078266793,
      "fallback_charge_usd": 0.0003335,
      "cost_savings_fraction": 0.27573179720916596,
      "accuracy_change_pp": -0.8000000000000007,
      "paired_correctness": {
        "gained": 3,
        "lost": 7,
        "unchanged_correctness": 490
      },
      "paid_tokens_in_replay": {
        "prompt_tokens": 1274,
        "completion_tokens": 10
      }
    },
    {
      "model": "qwen/qwen3-8b",
      "name": "Qwen: Qwen3 8B",
      "direct": {
        "n": 500,
        "correct": 496,
        "accuracy": 0.992,
        "accuracy_wilson_95": [
          0.9796129641658948,
          0.9968846848199377
        ],
        "total_seconds": 332.8775847758516,
        "median_seconds": 0.645228125504218,
        "p95_seconds": 0.8494925000122748
      },
      "hybrid_replay": {
        "n": 500,
        "correct": 491,
        "accuracy": 0.982,
        "accuracy_wilson_95": [
          0.9661483290912607,
          0.990501806703803
        ],
        "total_seconds": 89.75026233983226,
        "median_seconds": 0.1622084794798866,
        "p95_seconds": 0.22851016698405147
      },
      "local_accepted": 490,
      "local_accepted_errors": 8,
      "escalations": 10,
      "avoided_api_call_fraction": 0.98,
      "direct_cost_per_1000": 0.016677284,
      "hybrid_cost_per_1000": 0.02340873807826679,
      "fallback_charge_usd": 0.000169286,
      "cost_savings_fraction": -0.40363011616680455,
      "accuracy_change_pp": -1.0000000000000009,
      "paired_correctness": {
        "gained": 2,
        "lost": 7,
        "unchanged_correctness": 491
      },
      "paid_tokens_in_replay": {
        "prompt_tokens": 1408,
        "completion_tokens": 10
      }
    },
    {
      "model": "google/gemma-3-27b-it",
      "name": "Google: Gemma 3 27B",
      "direct": {
        "n": 500,
        "correct": 496,
        "accuracy": 0.992,
        "accuracy_wilson_95": [
          0.9796129641658948,
          0.9968846848199377
        ],
        "total_seconds": 161.43429854873102,
        "median_seconds": 0.29188295849598944,
        "p95_seconds": 0.5982790420530364
      },
      "hybrid_replay": {
        "n": 500,
        "correct": 492,
        "accuracy": 0.984,
        "accuracy_wilson_95": [
          0.9687488914666058,
          0.9918707469666117
        ],
        "total_seconds": 86.20800538395997,
        "median_seconds": 0.1622084794798866,
        "p95_seconds": 0.22851016698405147
      },
      "local_accepted": 490,
      "local_accepted_errors": 8,
      "escalations": 10,
      "avoided_api_call_fraction": 0.98,
      "direct_cost_per_1000": 0.0141684,
      "hybrid_cost_per_1000": 0.023358766078266792,
      "fallback_charge_usd": 0.0001443,
      "cost_savings_fraction": -0.6486523586478921,
      "accuracy_change_pp": -0.8000000000000007,
      "paired_correctness": {
        "gained": 3,
        "lost": 7,
        "unchanged_correctness": 490
      },
      "paid_tokens_in_replay": {
        "prompt_tokens": 1374,
        "completion_tokens": 23
      }
    },
    {
      "model": "meta-llama/llama-3.2-1b-instruct",
      "name": "Meta: Llama 3.2 1B Instruct",
      "direct": {
        "n": 500,
        "correct": 49,
        "accuracy": 0.098,
        "accuracy_wilson_95": [
          0.07492391886460623,
          0.12720605086648182
        ],
        "total_seconds": 1800.5567155783065,
        "median_seconds": 2.3418112919898704,
        "p95_seconds": 9.874993375036865
      },
      "hybrid_replay": {
        "n": 500,
        "correct": 483,
        "accuracy": 0.966,
        "accuracy_wilson_95": [
          0.9462286344983655,
          0.9786654801914678
        ],
        "total_seconds": 124.45412929781014,
        "median_seconds": 0.1622084794798866,
        "p95_seconds": 0.22851016698405147
      },
      "local_accepted": 490,
      "local_accepted_errors": 8,
      "escalations": 10,
      "avoided_api_call_fraction": 0.98,
      "direct_cost_per_1000": 0.008577138,
      "hybrid_cost_per_1000": 0.023242378078266793,
      "fallback_charge_usd": 8.6106e-05,
      "cost_savings_fraction": -1.7098057741716168,
      "accuracy_change_pp": 86.8,
      "paired_correctness": {
        "gained": 434,
        "lost": 0,
        "unchanged_correctness": 66
      },
      "paid_tokens_in_replay": {
        "prompt_tokens": 1395,
        "completion_tokens": 241
      }
    },
    {
      "model": "meta-llama/llama-3.2-3b-instruct",
      "name": "Meta: Llama 3.2 3B Instruct",
      "direct": {
        "n": 500,
        "correct": 439,
        "accuracy": 0.878,
        "accuracy_wilson_95": [
          0.8463952769261602,
          0.9038407216849063
        ],
        "total_seconds": 124.94482434284873,
        "median_seconds": 0.22435187551309355,
        "p95_seconds": 0.39086033403873444
      },
      "hybrid_replay": {
        "n": 500,
        "correct": 491,
        "accuracy": 0.982,
        "accuracy_wilson_95": [
          0.9661483290912607,
          0.990501806703803
        ],
        "total_seconds": 85.26503138273256,
        "median_seconds": 0.1622084794798866,
        "p95_seconds": 0.22851016698405147
      },
      "local_accepted": 490,
      "local_accepted_errors": 8,
      "escalations": 10,
      "avoided_api_call_fraction": 0.98,
      "direct_cost_per_1000": 0.0087687208,
      "hybrid_cost_per_1000": 0.023245589078266793,
      "fallback_charge_usd": 8.77115e-05,
      "cost_savings_fraction": -1.6509669549823953,
      "accuracy_change_pp": 10.399999999999999,
      "paired_correctness": {
        "gained": 55,
        "lost": 3,
        "unchanged_correctness": 442
      },
      "paid_tokens_in_replay": {
        "prompt_tokens": 1585,
        "completion_tokens": 21
      }
    },
    {
      "model": "meta-llama/llama-3.1-8b-instruct",
      "name": "Meta: Llama 3.1 8B Instruct",
      "direct": {
        "n": 500,
        "correct": 480,
        "accuracy": 0.96,
        "accuracy_wilson_95": [
          0.9390264041651492,
          0.9739592026102227
        ],
        "total_seconds": 59.26007748604752,
        "median_seconds": 0.09963452053489164,
        "p95_seconds": 0.15847958397353068
      },
      "hybrid_replay": {
        "n": 500,
        "correct": 492,
        "accuracy": 0.984,
        "accuracy_wilson_95": [
          0.9687488914666058,
          0.9918707469666117
        ],
        "total_seconds": 84.19550363073358,
        "median_seconds": 0.1622084794798866,
        "p95_seconds": 0.22607908298959956
      },
      "local_accepted": 490,
      "local_accepted_errors": 8,
      "escalations": 10,
      "avoided_api_call_fraction": 0.98,
      "direct_cost_per_1000": 0.03068516,
      "hybrid_cost_per_1000": 0.023692766078266793,
      "fallback_charge_usd": 0.0003113,
      "cost_savings_fraction": 0.22787542648411174,
      "accuracy_change_pp": 2.400000000000002,
      "paired_correctness": {
        "gained": 17,
        "lost": 5,
        "unchanged_correctness": 478
      },
      "paid_tokens_in_replay": {
        "prompt_tokens": 1395,
        "completion_tokens": 20
      }
    },
    {
      "model": "meta-llama/llama-3.3-70b-instruct",
      "name": "Meta: Llama 3.3 70B Instruct",
      "direct": {
        "n": 500,
        "correct": 494,
        "accuracy": 0.988,
        "accuracy_wilson_95": [
          0.9740696388835527,
          0.9944890048259724
        ],
        "total_seconds": 317.2056205851841,
        "median_seconds": 0.4161451875115745,
        "p95_seconds": 1.668889874999877
      },
      "hybrid_replay": {
        "n": 500,
        "correct": 492,
        "accuracy": 0.984,
        "accuracy_wilson_95": [
          0.9687488914666058,
          0.9918707469666117
        ],
        "total_seconds": 87.44755342369899,
        "median_seconds": 0.1622084794798866,
        "p95_seconds": 0.22851016698405147
      },
      "local_accepted": 490,
      "local_accepted_errors": 8,
      "escalations": 10,
      "avoided_api_call_fraction": 0.98,
      "direct_cost_per_1000": 0.0297207,
      "hybrid_cost_per_1000": 0.023596966078266793,
      "fallback_charge_usd": 0.0002634,
      "cost_savings_fraction": 0.20604272179771022,
      "accuracy_change_pp": -0.40000000000000036,
      "paired_correctness": {
        "gained": 5,
        "lost": 7,
        "unchanged_correctness": 488
      },
      "paid_tokens_in_replay": {
        "prompt_tokens": 1585,
        "completion_tokens": 20
      }
    },
    {
      "model": "meta-llama/llama-4-maverick",
      "name": "Meta: Llama 4 Maverick",
      "direct": {
        "n": 500,
        "correct": 494,
        "accuracy": 0.988,
        "accuracy_wilson_95": [
          0.9740696388835527,
          0.9944890048259724
        ],
        "total_seconds": 317.1911564466427,
        "median_seconds": 0.5745751454960555,
        "p95_seconds": 0.9321282090386376
      },
      "hybrid_replay": {
        "n": 500,
        "correct": 492,
        "accuracy": 0.984,
        "accuracy_wilson_95": [
          0.9687488914666058,
          0.9918707469666117
        ],
        "total_seconds": 89.57759958982933,
        "median_seconds": 0.1622084794798866,
        "p95_seconds": 0.22851016698405147
      },
      "local_accepted": 490,
      "local_accepted_errors": 8,
      "escalations": 10,
      "avoided_api_call_fraction": 0.98,
      "direct_cost_per_1000": 0.02669025,
      "hybrid_cost_per_1000": 0.023611891078266792,
      "fallback_charge_usd": 0.0002708625,
      "cost_savings_fraction": 0.11533645888416955,
      "accuracy_change_pp": -0.40000000000000036,
      "paired_correctness": {
        "gained": 4,
        "lost": 6,
        "unchanged_correctness": 490
      },
      "paid_tokens_in_replay": {
        "prompt_tokens": 1375,
        "completion_tokens": 20
      }
    },
    {
      "model": "mistralai/mistral-nemo",
      "name": "Mistral: Mistral Nemo",
      "direct": {
        "n": 500,
        "correct": 483,
        "accuracy": 0.966,
        "accuracy_wilson_95": [
          0.9462286344983655,
          0.9786654801914678
        ],
        "total_seconds": 191.0904435516568,
        "median_seconds": 0.3122280004899949,
        "p95_seconds": 0.5505917079863138
      },
      "hybrid_replay": {
        "n": 500,
        "correct": 491,
        "accuracy": 0.982,
        "accuracy_wilson_95": [
          0.9661483290912607,
          0.990501806703803
        ],
        "total_seconds": 87.98313083988614,
        "median_seconds": 0.1622084794798866,
        "p95_seconds": 0.22851016698405147
      },
      "local_accepted": 490,
      "local_accepted_errors": 8,
      "escalations": 10,
      "avoided_api_call_fraction": 0.98,
      "direct_cost_per_1000": 0.005514016,
      "hybrid_cost_per_1000": 0.02318008607826679,
      "fallback_charge_usd": 5.496e-05,
      "cost_savings_fraction": -3.203848171326814,
      "accuracy_change_pp": 1.6000000000000014,
      "paired_correctness": {
        "gained": 14,
        "lost": 6,
        "unchanged_correctness": 480
      },
      "paid_tokens_in_replay": {
        "prompt_tokens": 1339,
        "completion_tokens": 26
      }
    },
    {
      "model": "mistralai/mistral-small-3.2-24b-instruct",
      "name": "Mistral: Mistral Small 3.2 24B",
      "direct": {
        "n": 500,
        "correct": 477,
        "accuracy": 0.954,
        "accuracy_wilson_95": [
          0.9319222072317267,
          0.9691548916291838
        ],
        "total_seconds": 493.9661440376076,
        "median_seconds": 0.6996111879998352,
        "p95_seconds": 2.9433915420086123
      },
      "hybrid_replay": {
        "n": 500,
        "correct": 491,
        "accuracy": 0.982,
        "accuracy_wilson_95": [
          0.9661483290912607,
          0.990501806703803
        ],
        "total_seconds": 94.26939721481176,
        "median_seconds": 0.1622084794798866,
        "p95_seconds": 0.22851016698405147
      },
      "local_accepted": 490,
      "local_accepted_errors": 8,
      "escalations": 10,
      "avoided_api_call_fraction": 0.98,
      "direct_cost_per_1000": null,
      "hybrid_cost_per_1000": 0.02333460357826679,
      "fallback_charge_usd": 0.00013221875,
      "cost_savings_fraction": null,
      "accuracy_change_pp": 2.8000000000000025,
      "paired_correctness": {
        "gained": 20,
        "lost": 6,
        "unchanged_correctness": 474
      },
      "paid_tokens_in_replay": {
        "prompt_tokens": 1357,
        "completion_tokens": 20
      }
    },
    {
      "model": "mistralai/ministral-3b-2512",
      "name": "Mistral: Ministral 3 3B 2512",
      "direct": {
        "n": 500,
        "correct": 457,
        "accuracy": 0.914,
        "accuracy_wilson_95": [
          0.8861601885014425,
          0.9355268575963924
        ],
        "total_seconds": 207.9254797201138,
        "median_seconds": 0.3732963960210327,
        "p95_seconds": 0.6952591249719262
      },
      "hybrid_replay": {
        "n": 500,
        "correct": 490,
        "accuracy": 0.98,
        "accuracy_wilson_95": [
          0.9635798167531208,
          0.9891008164037892
        ],
        "total_seconds": 88.85190517577576,
        "median_seconds": 0.1622084794798866,
        "p95_seconds": 0.22851016698405147
      },
      "local_accepted": 490,
      "local_accepted_errors": 8,
      "escalations": 10,
      "avoided_api_call_fraction": 0.98,
      "direct_cost_per_1000": 0.01212064,
      "hybrid_cost_per_1000": 0.02327644607826679,
      "fallback_charge_usd": 0.00010314,
      "cost_savings_fraction": -0.9203974442163774,
      "accuracy_change_pp": 6.599999999999994,
      "paired_correctness": {
        "gained": 36,
        "lost": 3,
        "unchanged_correctness": 461
      },
      "paid_tokens_in_replay": {
        "prompt_tokens": 1357,
        "completion_tokens": 20
      }
    },
    {
      "model": "mistralai/mistral-large-2512",
      "name": "Mistral: Mistral Large 3 2512",
      "direct": {
        "n": 500,
        "correct": 497,
        "accuracy": 0.994,
        "accuracy_wilson_95": [
          0.9825097478959467,
          0.9979574037280398
        ],
        "total_seconds": 349.18739466846455,
        "median_seconds": 0.628262458514655,
        "p95_seconds": 1.1614113749819808
      },
      "hybrid_replay": {
        "n": 500,
        "correct": 492,
        "accuracy": 0.984,
        "accuracy_wilson_95": [
          0.9687488914666058,
          0.9918707469666117
        ],
        "total_seconds": 89.7949872147874,
        "median_seconds": 0.1622084794798866,
        "p95_seconds": 0.22851016698405147
      },
      "local_accepted": 490,
      "local_accepted_errors": 8,
      "escalations": 10,
      "avoided_api_call_fraction": 0.98,
      "direct_cost_per_1000": 0.069439,
      "hybrid_cost_per_1000": 0.024487166078266794,
      "fallback_charge_usd": 0.0007085,
      "cost_savings_fraction": 0.647357161274402,
      "accuracy_change_pp": -1.0000000000000009,
      "paired_correctness": {
        "gained": 2,
        "lost": 7,
        "unchanged_correctness": 491
      },
      "paid_tokens_in_replay": {
        "prompt_tokens": 1357,
        "completion_tokens": 20
      }
    },
    {
      "model": "qwen/qwen3.7-flash",
      "name": "Qwen: Qwen3.7 Flash",
      "direct": {
        "n": 500,
        "correct": 483,
        "accuracy": 0.966,
        "accuracy_wilson_95": [
          0.9462286344983655,
          0.9786654801914678
        ],
        "total_seconds": 427.76769975124625,
        "median_seconds": 0.6487369170063175,
        "p95_seconds": 1.1781788329826668
      },
      "hybrid_replay": {
        "n": 500,
        "correct": 492,
        "accuracy": 0.984,
        "accuracy_wilson_95": [
          0.9687488914666058,
          0.9918707469666117
        ],
        "total_seconds": 89.78811667481204,
        "median_seconds": 0.1622084794798866,
        "p95_seconds": 0.22851016698405147
      },
      "local_accepted": 490,
      "local_accepted_errors": 8,
      "escalations": 10,
      "avoided_api_call_fraction": 0.98,
      "direct_cost_per_1000": null,
      "hybrid_cost_per_1000": 0.023157186078266792,
      "fallback_charge_usd": 4.351e-05,
      "cost_savings_fraction": null,
      "accuracy_change_pp": 1.8000000000000016,
      "paired_correctness": {
        "gained": 14,
        "lost": 5,
        "unchanged_correctness": 481
      },
      "paid_tokens_in_replay": {
        "prompt_tokens": 1407,
        "completion_tokens": 10
      }
    },
    {
      "model": "qwen/qwen3.8-flash",
      "name": "Qwen: Qwen3.8 Flash",
      "direct": {
        "n": 500,
        "correct": 495,
        "accuracy": 0.99,
        "accuracy_wilson_95": [
          0.9768069002442693,
          0.9957212461034096
        ],
        "total_seconds": 530.1021604596172,
        "median_seconds": 0.7433171665179543,
        "p95_seconds": 2.022013458015863
      },
      "hybrid_replay": {
        "n": 500,
        "correct": 491,
        "accuracy": 0.982,
        "accuracy_wilson_95": [
          0.9661483290912607,
          0.990501806703803
        ],
        "total_seconds": 100.04612713080132,
        "median_seconds": 0.1622084794798866,
        "p95_seconds": 0.22851016698405147
      },
      "local_accepted": 490,
      "local_accepted_errors": 8,
      "escalations": 10,
      "avoided_api_call_fraction": 0.98,
      "direct_cost_per_1000": null,
      "hybrid_cost_per_1000": 0.023501666078266793,
      "fallback_charge_usd": 0.00021575,
      "cost_savings_fraction": null,
      "accuracy_change_pp": -0.8000000000000007,
      "paired_correctness": {
        "gained": 3,
        "lost": 7,
        "unchanged_correctness": 490
      },
      "paid_tokens_in_replay": {
        "prompt_tokens": 1407,
        "completion_tokens": 10
      }
    },
    {
      "model": "qwen/qwen3-30b-a3b-instruct-2507",
      "name": "Qwen: Qwen3 30B A3B Instruct 2507",
      "direct": {
        "n": 500,
        "correct": 495,
        "accuracy": 0.99,
        "accuracy_wilson_95": [
          0.9768069002442693,
          0.9957212461034096
        ],
        "total_seconds": 972.880771335389,
        "median_seconds": 1.8299474790110253,
        "p95_seconds": 3.2622302920208313
      },
      "hybrid_replay": {
        "n": 500,
        "correct": 492,
        "accuracy": 0.984,
        "accuracy_wilson_95": [
          0.9687488914666058,
          0.9918707469666117
        ],
        "total_seconds": 111.89126233878778,
        "median_seconds": 0.1622084794798866,
        "p95_seconds": 0.22851016698405147
      },
      "local_accepted": 490,
      "local_accepted_errors": 8,
      "escalations": 10,
      "avoided_api_call_fraction": 0.98,
      "direct_cost_per_1000": 0.0084885357,
      "hybrid_cost_per_1000": 0.023220451378266792,
      "fallback_charge_usd": 7.514265e-05,
      "cost_savings_fraction": -1.7355073005426354,
      "accuracy_change_pp": -0.6000000000000005,
      "paired_correctness": {
        "gained": 3,
        "lost": 6,
        "unchanged_correctness": 491
      },
      "paid_tokens_in_replay": {
        "prompt_tokens": 1368,
        "completion_tokens": 18
      }
    },
    {
      "model": "deepseek/deepseek-v4-flash-0731",
      "name": "DeepSeek: DeepSeek V4 Flash 0731",
      "direct": {
        "n": 500,
        "correct": 491,
        "accuracy": 0.982,
        "accuracy_wilson_95": [
          0.9661483290912607,
          0.990501806703803
        ],
        "total_seconds": 726.8716526252683,
        "median_seconds": 0.5665040830208454,
        "p95_seconds": 5.947828042029869
      },
      "hybrid_replay": {
        "n": 500,
        "correct": 492,
        "accuracy": 0.984,
        "accuracy_wilson_95": [
          0.9687488914666058,
          0.9918707469666117
        ],
        "total_seconds": 106.2602106318227,
        "median_seconds": 0.1622084794798866,
        "p95_seconds": 0.22851016698405147
      },
      "local_accepted": 490,
      "local_accepted_errors": 8,
      "escalations": 10,
      "avoided_api_call_fraction": 0.98,
      "direct_cost_per_1000": 0.0080208628,
      "hybrid_cost_per_1000": 0.02323396607826679,
      "fallback_charge_usd": 8.19e-05,
      "cost_savings_fraction": -1.8966916225355193,
      "accuracy_change_pp": 0.20000000000000018,
      "paired_correctness": {
        "gained": 7,
        "lost": 6,
        "unchanged_correctness": 487
      },
      "paid_tokens_in_replay": {
        "prompt_tokens": 1325,
        "completion_tokens": 20
      }
    },
    {
      "model": "amazon/nova-micro-v1",
      "name": "Amazon: Nova Micro 1.0",
      "direct": {
        "n": 500,
        "correct": 491,
        "accuracy": 0.982,
        "accuracy_wilson_95": [
          0.9661483290912607,
          0.990501806703803
        ],
        "total_seconds": 210.47532196046086,
        "median_seconds": 0.4090766459994484,
        "p95_seconds": 0.5453609160031192
      },
      "hybrid_replay": {
        "n": 500,
        "correct": 492,
        "accuracy": 0.984,
        "accuracy_wilson_95": [
          0.9687488914666058,
          0.9918707469666117
        ],
        "total_seconds": 87.8335815887549,
        "median_seconds": 0.1622084794798866,
        "p95_seconds": 0.22851016698405147
      },
      "local_accepted": 490,
      "local_accepted_errors": 8,
      "escalations": 10,
      "avoided_api_call_fraction": 0.98,
      "direct_cost_per_1000": 0.00501844,
      "hybrid_cost_per_1000": 0.023173486078266792,
      "fallback_charge_usd": 5.166e-05,
      "cost_savings_fraction": -3.6176672588028937,
      "accuracy_change_pp": 0.20000000000000018,
      "paired_correctness": {
        "gained": 6,
        "lost": 5,
        "unchanged_correctness": 489
      },
      "paid_tokens_in_replay": {
        "prompt_tokens": 1364,
        "completion_tokens": 28
      }
    },
    {
      "model": "amazon/nova-lite-v1",
      "name": "Amazon: Nova Lite 1.0",
      "direct": {
        "n": 500,
        "correct": 493,
        "accuracy": 0.986,
        "accuracy_wilson_95": [
          0.9713869308622147,
          0.9932022102091564
        ],
        "total_seconds": 410.10159253078746,
        "median_seconds": 0.4755054999841377,
        "p95_seconds": 2.969197292055469
      },
      "hybrid_replay": {
        "n": 500,
        "correct": 492,
        "accuracy": 0.984,
        "accuracy_wilson_95": [
          0.9687488914666058,
          0.9918707469666117
        ],
        "total_seconds": 87.82772663177457,
        "median_seconds": 0.1622084794798866,
        "p95_seconds": 0.22851016698405147
      },
      "local_accepted": 490,
      "local_accepted_errors": 8,
      "escalations": 10,
      "avoided_api_call_fraction": 0.98,
      "direct_cost_per_1000": 0.00859776,
      "hybrid_cost_per_1000": 0.023248246078266793,
      "fallback_charge_usd": 8.904e-05,
      "cost_savings_fraction": -1.7039887224424493,
      "accuracy_change_pp": -0.20000000000000018,
      "paired_correctness": {
        "gained": 6,
        "lost": 7,
        "unchanged_correctness": 487
      },
      "paid_tokens_in_replay": {
        "prompt_tokens": 1364,
        "completion_tokens": 30
      }
    },
    {
      "model": "cohere/command-r7b-12-2024",
      "name": "Cohere: Command R7B (12-2024)",
      "direct": {
        "n": 500,
        "correct": 485,
        "accuracy": 0.97,
        "accuracy_wilson_95": [
          0.9510963331640543,
          0.9817367868020865
        ],
        "total_seconds": 121.57807346608024,
        "median_seconds": 0.22833731249556877,
        "p95_seconds": 0.31850850000046194
      },
      "hybrid_replay": {
        "n": 500,
        "correct": 491,
        "accuracy": 0.982,
        "accuracy_wilson_95": [
          0.9661483290912607,
          0.990501806703803
        ],
        "total_seconds": 85.42226204869803,
        "median_seconds": 0.1622084794798866,
        "p95_seconds": 0.22851016698405147
      },
      "local_accepted": 490,
      "local_accepted_errors": 8,
      "escalations": 10,
      "avoided_api_call_fraction": 0.98,
      "direct_cost_per_1000": 0.0048249,
      "hybrid_cost_per_1000": 0.023168416078266793,
      "fallback_charge_usd": 4.9125e-05,
      "cost_savings_fraction": -3.8018437850042055,
      "accuracy_change_pp": 1.200000000000001,
      "paired_correctness": {
        "gained": 10,
        "lost": 4,
        "unchanged_correctness": 486
      },
      "paid_tokens_in_replay": {
        "prompt_tokens": 1270,
        "completion_tokens": 10
      }
    },
    {
      "model": "nvidia/nemotron-3.5-lightning",
      "name": "NVIDIA: Nemotron 3.5 Lightning",
      "direct": {
        "n": 500,
        "correct": 480,
        "accuracy": 0.96,
        "accuracy_wilson_95": [
          0.9390264041651492,
          0.9739592026102227
        ],
        "total_seconds": 75.26812473102473,
        "median_seconds": 0.13728989599621855,
        "p95_seconds": 0.22562079096678644
      },
      "hybrid_replay": {
        "n": 500,
        "correct": 492,
        "accuracy": 0.984,
        "accuracy_wilson_95": [
          0.9687488914666058,
          0.9918707469666117
        ],
        "total_seconds": 84.37686667469097,
        "median_seconds": 0.1622084794798866,
        "p95_seconds": 0.22851016698405147
      },
      "local_accepted": 490,
      "local_accepted_errors": 8,
      "escalations": 10,
      "avoided_api_call_fraction": 0.98,
      "direct_cost_per_1000": 0.0105051,
      "hybrid_cost_per_1000": 0.023284346078266793,
      "fallback_charge_usd": 0.00010709,
      "cost_savings_fraction": -1.2164801932648706,
      "accuracy_change_pp": 2.400000000000002,
      "paired_correctness": {
        "gained": 18,
        "lost": 6,
        "unchanged_correctness": 476
      },
      "paid_tokens_in_replay": {
        "prompt_tokens": 1467,
        "completion_tokens": 22
      }
    }
  ]
}
