{
  "completed_at": "2026-09-17T03:45:30.552057+00:00",
  "scope": "Three requested leading models at low reasoning effort on the same 500 eight-choice CLINC routing cases. Exploratory extension of a previously examined test. Direct API outcomes are measured; local-gate results replay earlier warm local records. Not a live hybrid, a general intelligence ranking, or a maximum-reasoning comparison.",
  "source_evidence_manifest_sha256": "18f74fedc18d6e7b43c175644df8833e9ceb31a3e696d73f071cc04401d8095e",
  "report_tool_sha256": "ee15df940b03f8aa255af904795645a7f19d5e710b12f3fbff3079840ab29b5e",
  "threshold_selection": {
    "threshold": 0.8502035140991211,
    "accepted": 193,
    "correct": 192,
    "validation_n": 200
  },
  "local_cost_assumptions": {
    "allocated_hourly_usd": 0.5,
    "utilization": 1
  },
  "local_hardware": "Apple M4 Max, MPS, torch 2.8.0; warm sequential inference",
  "client_conditions": "The workstation also ran brief local ERP development probes during part of this API run. API timing is observed client latency, not an isolated hardware benchmark; local routing timings are reused from the earlier run.",
  "local": {
    "model": "openauditor/v0.3.0",
    "name": "OpenAuditor 0.6B",
    "kind": "local",
    "measurement": {
      "n": 500,
      "correct": 486,
      "accuracy": 0.972,
      "accuracy_wilson_95": [
        0.9535535973250511,
        0.9832490252792436
      ],
      "total_seconds": 83.05259788176045,
      "median_seconds": 0.16062327049439773,
      "p95_seconds": 0.22153566702036187
    },
    "cost_per_1000": 0.023070166078266792,
    "cost_basis": "Scenario: $0.50 per allocated machine-hour, 100% utilization",
    "errors": {},
    "numeric_key_compliance": null
  },
  "entries": [
    {
      "model": "anthropic/claude-fable-5",
      "name": "Claude Fable 5",
      "direct": {
        "n": 500,
        "correct": 497,
        "accuracy": 0.994,
        "accuracy_wilson_95": [
          0.9825097478959467,
          0.9979574037280398
        ],
        "total_seconds": 1669.5903178428998,
        "median_seconds": 3.208765791991027,
        "p95_seconds": 4.232279624964576
      },
      "hybrid_replay": {
        "n": 500,
        "correct": 491,
        "accuracy": 0.982,
        "accuracy_wilson_95": [
          0.9661483290912607,
          0.990501806703803
        ],
        "total_seconds": 115.70262513088528,
        "median_seconds": 0.1622084794798866,
        "p95_seconds": 0.22851016698405147
      },
      "local_accepted": 490,
      "local_accepted_errors": 8,
      "escalations": 10,
      "avoided_api_call_fraction": 0.98,
      "direct_cost_per_1000": 1.9394,
      "hybrid_cost_per_1000": 0.061690166078266793,
      "fallback_charge_usd": 0.01931,
      "cost_savings_fraction": 0.9681911075186826,
      "accuracy_change_pp": -1.200000000000001,
      "paired_correctness": {
        "gained": 1,
        "lost": 7,
        "unchanged_correctness": 492
      },
      "paid_tokens_in_replay": {
        "prompt_tokens": 1781,
        "completion_tokens": 30
      },
      "median_speed_ratio": 19.781738921909476,
      "errors": {},
      "providers": {
        "Anthropic": 499,
        "Azure": 1
      },
      "input_tokens": 87710,
      "output_tokens": 1852,
      "reasoning_tokens": 352,
      "unverified_charges": 0
    },
    {
      "model": "openai/gpt-6-astra",
      "name": "GPT-6 Astra",
      "direct": {
        "n": 500,
        "correct": 498,
        "accuracy": 0.996,
        "accuracy_wilson_95": [
          0.9855342847848231,
          0.9989023694773173
        ],
        "total_seconds": 768.3140909158392,
        "median_seconds": 1.3488265209889505,
        "p95_seconds": 2.720605082984548
      },
      "hybrid_replay": {
        "n": 500,
        "correct": 492,
        "accuracy": 0.984,
        "accuracy_wilson_95": [
          0.9687488914666058,
          0.9918707469666117
        ],
        "total_seconds": 97.61870542372344,
        "median_seconds": 0.1622084794798866,
        "p95_seconds": 0.22851016698405147
      },
      "local_accepted": 490,
      "local_accepted_errors": 8,
      "escalations": 10,
      "avoided_api_call_fraction": 0.98,
      "direct_cost_per_1000": 1.5839,
      "hybrid_cost_per_1000": 0.05475016607826679,
      "fallback_charge_usd": 0.01584,
      "cost_savings_fraction": 0.9654333189732516,
      "accuracy_change_pp": -1.200000000000001,
      "paired_correctness": {
        "gained": 1,
        "lost": 7,
        "unchanged_correctness": 492
      },
      "paid_tokens_in_replay": {
        "prompt_tokens": 1334,
        "completion_tokens": 50
      },
      "median_speed_ratio": 8.315388476076562,
      "errors": {},
      "providers": {
        "OpenAI": 500
      },
      "input_tokens": 65690,
      "output_tokens": 2701,
      "reasoning_tokens": 183,
      "unverified_charges": 0
    },
    {
      "model": "deepseek/deepseek-v4-pro-0813",
      "name": "DeepSeek V4 Pro (0813)",
      "direct": {
        "n": 500,
        "correct": 499,
        "accuracy": 0.998,
        "accuracy_wilson_95": [
          0.9887592932948532,
          0.9996468636054407
        ],
        "total_seconds": 2295.1692707003676,
        "median_seconds": 3.2944941879832186,
        "p95_seconds": 12.528947709011845
      },
      "hybrid_replay": {
        "n": 500,
        "correct": 492,
        "accuracy": 0.984,
        "accuracy_wilson_95": [
          0.9687488914666058,
          0.9918707469666117
        ],
        "total_seconds": 152.82215038174763,
        "median_seconds": 0.1622084794798866,
        "p95_seconds": 0.22851016698405147
      },
      "local_accepted": 490,
      "local_accepted_errors": 8,
      "escalations": 10,
      "avoided_api_call_fraction": 0.98,
      "direct_cost_per_1000": 0.4478047752,
      "hybrid_cost_per_1000": 0.03950505767826679,
      "fallback_charge_usd": 0.0082174458,
      "cost_savings_fraction": 0.911780624356623,
      "accuracy_change_pp": -1.4000000000000012,
      "paired_correctness": {
        "gained": 1,
        "lost": 8,
        "unchanged_correctness": 491
      },
      "paid_tokens_in_replay": {
        "prompt_tokens": 1325,
        "completion_tokens": 2352
      },
      "median_speed_ratio": 20.31024640972439,
      "errors": {},
      "providers": {
        "Ionstream": 94,
        "StreamLake": 405,
        "GMICloud": 1
      },
      "input_tokens": 64924,
      "output_tokens": 54570,
      "reasoning_tokens": 53569,
      "unverified_charges": 0
    }
  ],
  "actual_test_charge_usd": 1.9855523876,
  "actual_preflight_charge_usd": 0.013141279,
  "actual_pilot_charge_usd": 0.0147400192
}
