{
  "schema": "agent-public-bench@1",
  "generatedAt": "2026-10-07T00:00:00.000Z",
  "url": "https://agent.sasid.ai/benchmarks/model-head-to-head",
  "study": {
    "slug": "model-head-to-head",
    "title": "Haiku vs Sonnet vs Opus vs Fable vs Codex: a timed head-to-head",
    "seoTitle": "Haiku vs Sonnet vs Opus vs Fable vs Codex: speed and tokens",
    "description": "130 timed calls on five validated tasks: Claude Haiku, Sonnet, Opus and Fable against Codex GPT-6.1 Sol. Pass rate, speed, tokens, cost per pass.",
    "question": "On short tasks with strict validators, how do the Claude Code models and efforts compare with Codex on pass rate, speed and tokens?",
    "answer": "127 of 130 calls passed (98%, 95% interval 93% to 99%), so on these short tasks pass rate barely separates the configurations (non-passes: Claude Sonnet 5.5 · Claude Code on Multi-step shift arithmetic ×3 (\"Output does not exactly match the expected text\")). Every non-pass ended on the expected answer (2292) but added working lines, which the exact-text validator rejects by design: a format miss, not a wrong answer. Speed separates them more: the fastest configuration was Claude Fable 5.1 · Claude Code at a median 1.9 s per call, the slowest GPT-6.1 Sol (low) · Codex CLI at 6.3 s. Claude Code medians ran from 1.9 s to 4.4 s. Codex CLI medians ran from 5.6 s to 6.3 s. Single calls vary a lot (see the ranges), so neighbouring configurations are not separated. Codex CLI sends a median 12,124 input tokens per call against 2,130 for Claude Code, mostly the CLI’s own context. At list price (a calculation; the calls ran on subscriptions), the cheapest passing answer came from Claude Sonnet 5.5 · Claude Code at $0.0062. On speed, cost per pass and pass rate together, the frontier is Claude Fable 5.1 · Claude Code, Claude Sonnet 5.5 · Claude Code, Claude Opus 5.5 (high) · Claude Code, Claude Opus 5.5 · Claude Code, Claude Opus 5.5 (low) · Claude Code; with 10 to 15 calls per configuration, small gaps inside it are within the spread of single calls. A harder follow-up with eight tasks and strict validators: /benchmarks/hard-model-head-to-head.",
    "date": "2026-10-05",
    "updated": "2026-10-05",
    "tags": [
      "head-to-head",
      "claude-haiku",
      "claude-sonnet",
      "claude-opus",
      "claude-fable",
      "codex",
      "latency"
    ],
    "method": [
      "Protocol declared before the first call: 5 cases with deterministic validators (behavioral checks in a sandbox, canonical JSON or exact text).",
      "Claude Code: Haiku, Sonnet, Opus and Fable at default effort, plus Opus at low and high, 3 repetitions. Codex: GPT-6.1 Sol at low, medium and high, 2 repetitions.",
      "Isolation: fresh empty working folder, tools off, no MCP servers, no session persistence, one turn. One call at a time per account.",
      "Every attempt is kept. Nothing is retried. A cell that did not run is shown as trimmed.",
      "All 130 planned calls ran. Nothing was trimmed or retried, and no run hit a usage or rate limit.",
      "Cost per passing answer: list price × reported tokens for every call in the configuration (cache reads and writes priced separately), divided by its passes."
    ],
    "caveats": [
      "The tasks are short and easy; pass rate saturates. Latency and tokens carry the signal. A harder follow-up with eight tasks and strict validators: /benchmarks/hard-model-head-to-head.",
      "Few repetitions per cell (2 or 3 per task). Medians with ranges, not intervals.",
      "CLI timings include CLI start-up and the CLI’s own system prompt.",
      "One host, one network, one day.",
      "List-price costs are calculations; the calls used flat subscriptions.",
      "The prompt cache stayed at the provider default, so cache counters differ by route and by call order.",
      "Haiku 4.5 reported reasoning tokens on most calls under the CLI default, which explains much of its extra time and output.",
      "Haiku and Fable are not in the platform runner catalog, so these calls used the runner’s CLI functions directly; each receipt records the model the CLI reported."
    ],
    "sourceIds": [
      "agent-provider-h2h",
      "calc-repricing",
      "price-anthropic",
      "price-openai"
    ],
    "stats": [
      {
        "id": "h2h-pass-all",
        "label": "Calls that passed their validator",
        "value": 0.9769,
        "unit": "rate",
        "display": "98% (127/130)",
        "n": 130,
        "ci": [
          0.9343,
          0.9921
        ]
      },
      {
        "id": "h2h-configs",
        "label": "Configurations compared",
        "value": 9,
        "unit": "count",
        "display": "9"
      },
      {
        "id": "h2h-fastest",
        "label": "Fastest configuration (median total time)",
        "value": 1.94,
        "unit": "seconds",
        "display": "Claude Fable 5.1 · Claude Code: 1.9 s",
        "n": 15
      },
      {
        "id": "h2h-slowest",
        "label": "Slowest configuration (median total time)",
        "value": 6.26,
        "unit": "seconds",
        "display": "GPT-6.1 Sol (low) · Codex CLI: 6.3 s",
        "n": 10
      },
      {
        "id": "h2h-cheapest-per-pass",
        "label": "Lowest list-price cost per passing answer (calculation)",
        "value": 0.00624,
        "unit": "usd",
        "display": "Claude Sonnet 5.5 · Claude Code: $0.0062",
        "n": 15
      },
      {
        "id": "h2h-codex-input-tokens",
        "label": "Median input tokens per call, Codex CLI vs Claude Code",
        "value": 12124,
        "unit": "tokens",
        "display": "12,124 vs 2,130",
        "n": 130
      }
    ],
    "charts": [
      {
        "id": "h2h-pass-rate",
        "title": "Pass rate on five validated tasks",
        "subtitle": "Every call counts; failures and timeouts are non-passes",
        "kind": "dot-range",
        "unit": "rate",
        "yLabel": "Passed",
        "series": [
          {
            "name": "Pass rate",
            "points": [
              {
                "label": "Claude Fable 5.1 · Claude Code",
                "value": 1,
                "lo": 0.7961,
                "hi": 1,
                "n": 15
              },
              {
                "label": "Claude Sonnet 5.5 · Claude Code",
                "value": 0.8,
                "lo": 0.5481,
                "hi": 0.9295,
                "n": 15
              },
              {
                "label": "Claude Opus 5.5 (high) · Claude Code",
                "value": 1,
                "lo": 0.7961,
                "hi": 1,
                "n": 15
              },
              {
                "label": "Claude Opus 5.5 · Claude Code",
                "value": 1,
                "lo": 0.7961,
                "hi": 1,
                "n": 15
              },
              {
                "label": "Claude Opus 5.5 (low) · Claude Code",
                "value": 1,
                "lo": 0.7961,
                "hi": 1,
                "n": 15
              },
              {
                "label": "Claude Haiku 4.5 · Claude Code",
                "value": 1,
                "lo": 0.7961,
                "hi": 1,
                "n": 15
              },
              {
                "label": "GPT-6.1 Sol (high) · Codex CLI",
                "value": 1,
                "lo": 0.7961,
                "hi": 1,
                "n": 15
              },
              {
                "label": "GPT-6.1 Sol (medium) · Codex CLI",
                "value": 1,
                "lo": 0.7961,
                "hi": 1,
                "n": 15
              },
              {
                "label": "GPT-6.1 Sol (low) · Codex CLI",
                "value": 1,
                "lo": 0.7225,
                "hi": 1,
                "n": 10
              }
            ]
          }
        ],
        "note": "Whiskers are 95% Wilson intervals. The tasks are short and most configurations pass nearly all of them, so pass rate does not separate the models here.",
        "sourceIds": [
          "agent-provider-h2h"
        ]
      },
      {
        "id": "h2h-total-latency",
        "title": "Total time per call",
        "subtitle": "Median per configuration; whiskers = fastest and slowest call",
        "kind": "dot-range",
        "unit": "seconds",
        "yLabel": "Seconds",
        "series": [
          {
            "name": "Total time per call",
            "points": [
              {
                "label": "Claude Fable 5.1 · Claude Code",
                "value": 1.94,
                "lo": 1.41,
                "hi": 9.83,
                "n": 15,
                "highlight": true
              },
              {
                "label": "Claude Sonnet 5.5 · Claude Code",
                "value": 2.31,
                "lo": 2.17,
                "hi": 7.73,
                "n": 15,
                "highlight": true
              },
              {
                "label": "Claude Opus 5.5 (high) · Claude Code",
                "value": 2.71,
                "lo": 2.45,
                "hi": 11.78,
                "n": 15,
                "highlight": true
              },
              {
                "label": "Claude Opus 5.5 · Claude Code",
                "value": 2.75,
                "lo": 2.47,
                "hi": 8.91,
                "n": 15,
                "highlight": true
              },
              {
                "label": "Claude Opus 5.5 (low) · Claude Code",
                "value": 2.83,
                "lo": 2.35,
                "hi": 6.62,
                "n": 15,
                "highlight": true
              },
              {
                "label": "Claude Haiku 4.5 · Claude Code",
                "value": 4.43,
                "lo": 3.16,
                "hi": 23.57,
                "n": 15,
                "highlight": true
              },
              {
                "label": "GPT-6.1 Sol (high) · Codex CLI",
                "value": 5.6,
                "lo": 4.05,
                "hi": 19.52,
                "n": 15,
                "highlight": false
              },
              {
                "label": "GPT-6.1 Sol (medium) · Codex CLI",
                "value": 5.65,
                "lo": 4.1,
                "hi": 25.46,
                "n": 15,
                "highlight": false
              },
              {
                "label": "GPT-6.1 Sol (low) · Codex CLI",
                "value": 6.26,
                "lo": 4.65,
                "hi": 10.47,
                "n": 10,
                "highlight": false
              }
            ]
          }
        ],
        "note": "One host, one network, one day. Whiskers are a range, not a confidence interval.",
        "sourceIds": [
          "agent-provider-h2h"
        ]
      },
      {
        "id": "h2h-first-useful-latency",
        "title": "Time to first useful output",
        "subtitle": "Median per configuration; whiskers = fastest and slowest call",
        "kind": "dot-range",
        "unit": "seconds",
        "yLabel": "Seconds",
        "series": [
          {
            "name": "Time to first useful output",
            "points": [
              {
                "label": "Claude Fable 5.1 · Claude Code",
                "value": 1.2,
                "lo": 0.95,
                "hi": 7.9,
                "n": 15,
                "highlight": true
              },
              {
                "label": "Claude Sonnet 5.5 · Claude Code",
                "value": 1.56,
                "lo": 0.99,
                "hi": 6.39,
                "n": 15,
                "highlight": true
              },
              {
                "label": "Claude Opus 5.5 (high) · Claude Code",
                "value": 2.04,
                "lo": 1.4,
                "hi": 9.94,
                "n": 15,
                "highlight": true
              },
              {
                "label": "Claude Opus 5.5 · Claude Code",
                "value": 1.92,
                "lo": 1.56,
                "hi": 7.23,
                "n": 15,
                "highlight": true
              },
              {
                "label": "Claude Opus 5.5 (low) · Claude Code",
                "value": 2.39,
                "lo": 1.45,
                "hi": 4.9,
                "n": 15,
                "highlight": true
              },
              {
                "label": "Claude Haiku 4.5 · Claude Code",
                "value": 3.63,
                "lo": 2.78,
                "hi": 22.27,
                "n": 15,
                "highlight": true
              },
              {
                "label": "GPT-6.1 Sol (high) · Codex CLI",
                "value": 5.32,
                "lo": 3.64,
                "hi": 16.37,
                "n": 15,
                "highlight": false
              },
              {
                "label": "GPT-6.1 Sol (medium) · Codex CLI",
                "value": 5.05,
                "lo": 3.36,
                "hi": 17.82,
                "n": 15,
                "highlight": false
              },
              {
                "label": "GPT-6.1 Sol (low) · Codex CLI",
                "value": 5.14,
                "lo": 4.02,
                "hi": 8.5,
                "n": 10,
                "highlight": false
              }
            ]
          }
        ],
        "note": "One host, one network, one day. Whiskers are a range, not a confidence interval.",
        "sourceIds": [
          "agent-provider-h2h"
        ]
      },
      {
        "id": "h2h-input-tokens",
        "title": "Input tokens per call: what the CLI sends",
        "subtitle": "Mean per call, split into prompt-cache reads and other input",
        "kind": "stacked-bar",
        "unit": "tokens",
        "yLabel": "Tokens",
        "series": [
          {
            "name": "Cache read",
            "points": [
              {
                "label": "Claude Fable 5.1 · Claude Code",
                "value": 2760,
                "n": 15
              },
              {
                "label": "Claude Sonnet 5.5 · Claude Code",
                "value": 1401,
                "n": 15
              },
              {
                "label": "Claude Opus 5.5 (high) · Claude Code",
                "value": 1463,
                "n": 15
              },
              {
                "label": "Claude Opus 5.5 · Claude Code",
                "value": 1401,
                "n": 15
              },
              {
                "label": "Claude Opus 5.5 (low) · Claude Code",
                "value": 1463,
                "n": 15
              },
              {
                "label": "Claude Haiku 4.5 · Claude Code",
                "value": 0,
                "n": 15
              },
              {
                "label": "GPT-6.1 Sol (high) · Codex CLI",
                "value": 6716,
                "n": 15
              },
              {
                "label": "GPT-6.1 Sol (medium) · Codex CLI",
                "value": 5180,
                "n": 15
              },
              {
                "label": "GPT-6.1 Sol (low) · Codex CLI",
                "value": 8064,
                "n": 10
              }
            ]
          },
          {
            "name": "Other input",
            "points": [
              {
                "label": "Claude Fable 5.1 · Claude Code",
                "value": 473,
                "n": 15
              },
              {
                "label": "Claude Sonnet 5.5 · Claude Code",
                "value": 685,
                "n": 15
              },
              {
                "label": "Claude Opus 5.5 (high) · Claude Code",
                "value": 619,
                "n": 15
              },
              {
                "label": "Claude Opus 5.5 · Claude Code",
                "value": 680,
                "n": 15
              },
              {
                "label": "Claude Opus 5.5 (low) · Claude Code",
                "value": 618,
                "n": 15
              },
              {
                "label": "Claude Haiku 4.5 · Claude Code",
                "value": 3790,
                "n": 15
              },
              {
                "label": "GPT-6.1 Sol (high) · Codex CLI",
                "value": 5406,
                "n": 15
              },
              {
                "label": "GPT-6.1 Sol (medium) · Codex CLI",
                "value": 6943,
                "n": 15
              },
              {
                "label": "GPT-6.1 Sol (low) · Codex CLI",
                "value": 4059,
                "n": 10
              }
            ]
          }
        ],
        "note": "The prompts are a few hundred tokens; most input is the CLI’s own system prompt and tool context. Input counts include cache reads, as the vendors report them.",
        "sourceIds": [
          "agent-provider-h2h"
        ]
      },
      {
        "id": "h2h-output-tokens",
        "title": "Output tokens per call",
        "subtitle": "Median per configuration; reasoning tokens where the CLI reports them",
        "kind": "grouped-bar",
        "unit": "tokens",
        "yLabel": "Tokens",
        "series": [
          {
            "name": "Output tokens",
            "points": [
              {
                "label": "Claude Fable 5.1 · Claude Code",
                "value": 64,
                "n": 15
              },
              {
                "label": "Claude Sonnet 5.5 · Claude Code",
                "value": 107,
                "n": 15
              },
              {
                "label": "Claude Opus 5.5 (high) · Claude Code",
                "value": 78,
                "n": 15
              },
              {
                "label": "Claude Opus 5.5 · Claude Code",
                "value": 64,
                "n": 15
              },
              {
                "label": "Claude Opus 5.5 (low) · Claude Code",
                "value": 64,
                "n": 15
              },
              {
                "label": "Claude Haiku 4.5 · Claude Code",
                "value": 367,
                "n": 15
              },
              {
                "label": "GPT-6.1 Sol (high) · Codex CLI",
                "value": 42,
                "n": 15
              },
              {
                "label": "GPT-6.1 Sol (medium) · Codex CLI",
                "value": 42,
                "n": 15
              },
              {
                "label": "GPT-6.1 Sol (low) · Codex CLI",
                "value": 42,
                "n": 10
              }
            ]
          },
          {
            "name": "Reasoning tokens",
            "points": [
              {
                "label": "Claude Fable 5.1 · Claude Code",
                "value": 0,
                "n": 15
              },
              {
                "label": "Claude Sonnet 5.5 · Claude Code",
                "value": 0,
                "n": 15
              },
              {
                "label": "Claude Opus 5.5 (high) · Claude Code",
                "value": 34,
                "n": 15
              },
              {
                "label": "Claude Opus 5.5 · Claude Code",
                "value": 0,
                "n": 15
              },
              {
                "label": "Claude Opus 5.5 (low) · Claude Code",
                "value": 0,
                "n": 15
              },
              {
                "label": "Claude Haiku 4.5 · Claude Code",
                "value": 297,
                "n": 15
              },
              {
                "label": "GPT-6.1 Sol (high) · Codex CLI",
                "value": 21,
                "n": 15
              },
              {
                "label": "GPT-6.1 Sol (medium) · Codex CLI",
                "value": 20,
                "n": 15
              },
              {
                "label": "GPT-6.1 Sol (low) · Codex CLI",
                "value": 20,
                "n": 10
              }
            ]
          }
        ],
        "note": "Vendors count reasoning differently; compare within a vendor first. A median of 0 can mean the CLI did not report reasoning for most calls.",
        "sourceIds": [
          "agent-provider-h2h"
        ]
      },
      {
        "id": "h2h-list-price-per-call",
        "title": "List-price cost per call (calculation)",
        "subtitle": "Reported tokens × list price; the calls ran on subscriptions",
        "kind": "dot-range",
        "unit": "usd",
        "yLabel": "USD per call",
        "series": [
          {
            "name": "Cost per call",
            "points": [
              {
                "label": "Claude Fable 5.1 · Claude Code",
                "value": 0.00987,
                "lo": 0.0049,
                "hi": 0.05843,
                "n": 15,
                "highlight": true
              },
              {
                "label": "Claude Sonnet 5.5 · Claude Code",
                "value": 0.0036,
                "lo": 0.00342,
                "hi": 0.01021,
                "n": 15,
                "highlight": true
              },
              {
                "label": "Claude Opus 5.5 (high) · Claude Code",
                "value": 0.00694,
                "lo": 0.00592,
                "hi": 0.02708,
                "n": 15,
                "highlight": true
              },
              {
                "label": "Claude Opus 5.5 · Claude Code",
                "value": 0.00688,
                "lo": 0.00592,
                "hi": 0.02226,
                "n": 15,
                "highlight": true
              },
              {
                "label": "Claude Opus 5.5 (low) · Claude Code",
                "value": 0.00688,
                "lo": 0.00582,
                "hi": 0.01793,
                "n": 15,
                "highlight": true
              },
              {
                "label": "Claude Haiku 4.5 · Claude Code",
                "value": 0.00566,
                "lo": 0.00513,
                "hi": 0.01804,
                "n": 15,
                "highlight": true
              },
              {
                "label": "GPT-6.1 Sol (high) · Codex CLI",
                "value": 0.01047,
                "lo": 0.0066,
                "hi": 0.02812,
                "n": 15,
                "highlight": false
              },
              {
                "label": "GPT-6.1 Sol (medium) · Codex CLI",
                "value": 0.01018,
                "lo": 0.0054,
                "hi": 0.02686,
                "n": 15,
                "highlight": false
              },
              {
                "label": "GPT-6.1 Sol (low) · Codex CLI",
                "value": 0.00769,
                "lo": 0.00742,
                "hi": 0.02649,
                "n": 10,
                "highlight": false
              }
            ]
          }
        ],
        "note": "Calculation, not a bill: the calls ran on flat subscriptions. Whiskers = cheapest and most expensive call.",
        "sourceIds": [
          "agent-provider-h2h",
          "calc-repricing",
          "price-anthropic",
          "price-openai"
        ]
      },
      {
        "id": "h2h-speed-vs-cost",
        "title": "Speed vs list-price cost",
        "subtitle": "Median total time and median list-price cost per call",
        "kind": "scatter",
        "unit": "seconds",
        "xLabel": "USD per call (list-price calculation)",
        "yLabel": "Median total seconds",
        "series": [
          {
            "name": "Configuration",
            "points": [
              {
                "label": "Claude Fable 5.1 · Claude Code",
                "x": 0.00987,
                "value": 1.94,
                "n": 15,
                "highlight": true
              },
              {
                "label": "Claude Sonnet 5.5 · Claude Code",
                "x": 0.0036,
                "value": 2.31,
                "n": 15,
                "highlight": true
              },
              {
                "label": "Claude Opus 5.5 (high) · Claude Code",
                "x": 0.00694,
                "value": 2.71,
                "n": 15,
                "highlight": true
              },
              {
                "label": "Claude Opus 5.5 · Claude Code",
                "x": 0.00688,
                "value": 2.75,
                "n": 15,
                "highlight": true
              },
              {
                "label": "Claude Opus 5.5 (low) · Claude Code",
                "x": 0.00688,
                "value": 2.83,
                "n": 15,
                "highlight": true
              },
              {
                "label": "Claude Haiku 4.5 · Claude Code",
                "x": 0.00566,
                "value": 4.43,
                "n": 15,
                "highlight": true
              },
              {
                "label": "GPT-6.1 Sol (high) · Codex CLI",
                "x": 0.01047,
                "value": 5.6,
                "n": 15,
                "highlight": false
              },
              {
                "label": "GPT-6.1 Sol (medium) · Codex CLI",
                "x": 0.01018,
                "value": 5.65,
                "n": 15,
                "highlight": false
              },
              {
                "label": "GPT-6.1 Sol (low) · Codex CLI",
                "x": 0.00769,
                "value": 6.26,
                "n": 10,
                "highlight": false
              }
            ]
          }
        ],
        "note": "Lower-left is faster and cheaper. Cost is a calculation from tokens.",
        "sourceIds": [
          "agent-provider-h2h",
          "calc-repricing",
          "price-anthropic",
          "price-openai"
        ]
      },
      {
        "id": "h2h-cost-per-pass",
        "title": "List-price cost per passing answer (calculation)",
        "subtitle": "All calls in a configuration, failures included, divided by its passes",
        "kind": "bar",
        "unit": "usd",
        "yLabel": "USD per passing answer",
        "series": [
          {
            "name": "Cost per pass",
            "points": [
              {
                "label": "Claude Sonnet 5.5 · Claude Code",
                "value": 0.00624,
                "n": 15,
                "highlight": true
              },
              {
                "label": "Claude Opus 5.5 (low) · Claude Code",
                "value": 0.00829,
                "n": 15,
                "highlight": true
              },
              {
                "label": "Claude Haiku 4.5 · Claude Code",
                "value": 0.00836,
                "n": 15,
                "highlight": false
              },
              {
                "label": "GPT-6.1 Sol (low) · Codex CLI",
                "value": 0.00998,
                "n": 10,
                "highlight": false
              },
              {
                "label": "Claude Opus 5.5 · Claude Code",
                "value": 0.01009,
                "n": 15,
                "highlight": true
              },
              {
                "label": "Claude Opus 5.5 (high) · Claude Code",
                "value": 0.01049,
                "n": 15,
                "highlight": true
              },
              {
                "label": "GPT-6.1 Sol (high) · Codex CLI",
                "value": 0.01322,
                "n": 15,
                "highlight": false
              },
              {
                "label": "GPT-6.1 Sol (medium) · Codex CLI",
                "value": 0.01564,
                "n": 15,
                "highlight": false
              },
              {
                "label": "Claude Fable 5.1 · Claude Code",
                "value": 0.02054,
                "n": 15,
                "highlight": true
              }
            ]
          }
        ],
        "note": "Calculation, not a bill: reported tokens × list price; the calls ran on flat subscriptions. A failed call still costs, so a lower pass rate raises the cost per pass. Highlighted bars are on the speed/cost/quality frontier.",
        "sourceIds": [
          "agent-provider-h2h",
          "calc-repricing",
          "price-anthropic",
          "price-openai"
        ]
      },
      {
        "id": "h2h-frontier",
        "title": "Speed, cost and quality frontier",
        "subtitle": "Median seconds against list-price cost per passing answer; pass rate in the note",
        "kind": "scatter",
        "unit": "seconds",
        "xLabel": "USD per passing answer (list-price calculation)",
        "yLabel": "Median total seconds",
        "series": [
          {
            "name": "Claude Code",
            "points": [
              {
                "label": "Claude Fable 5.1 · Claude Code",
                "x": 0.02054,
                "value": 1.94,
                "n": 15,
                "highlight": true
              },
              {
                "label": "Claude Sonnet 5.5 · Claude Code",
                "x": 0.00624,
                "value": 2.31,
                "n": 15,
                "highlight": true
              },
              {
                "label": "Claude Opus 5.5 (high) · Claude Code",
                "x": 0.01049,
                "value": 2.71,
                "n": 15,
                "highlight": true
              },
              {
                "label": "Claude Opus 5.5 · Claude Code",
                "x": 0.01009,
                "value": 2.75,
                "n": 15,
                "highlight": true
              },
              {
                "label": "Claude Opus 5.5 (low) · Claude Code",
                "x": 0.00829,
                "value": 2.83,
                "n": 15,
                "highlight": true
              },
              {
                "label": "Claude Haiku 4.5 · Claude Code",
                "x": 0.00836,
                "value": 4.43,
                "n": 15,
                "highlight": false
              }
            ]
          },
          {
            "name": "Codex CLI",
            "points": [
              {
                "label": "GPT-6.1 Sol (high) · Codex CLI",
                "x": 0.01322,
                "value": 5.6,
                "n": 15,
                "highlight": false
              },
              {
                "label": "GPT-6.1 Sol (medium) · Codex CLI",
                "x": 0.01564,
                "value": 5.65,
                "n": 15,
                "highlight": false
              },
              {
                "label": "GPT-6.1 Sol (low) · Codex CLI",
                "x": 0.00998,
                "value": 6.26,
                "n": 10,
                "highlight": false
              }
            ]
          }
        ],
        "note": "Lower-left is better. Highlighted points are on the frontier: no other configuration is at least as fast, as cheap per pass and as accurate. Frontier: Claude Fable 5.1 · Claude Code; Claude Sonnet 5.5 · Claude Code; Claude Opus 5.5 (high) · Claude Code; Claude Opus 5.5 · Claude Code; Claude Opus 5.5 (low) · Claude Code. Pass rates: Claude Fable 5.1 · Claude Code 15/15; Claude Sonnet 5.5 · Claude Code 12/15; Claude Opus 5.5 (high) · Claude Code 15/15; Claude Opus 5.5 · Claude Code 15/15; Claude Opus 5.5 (low) · Claude Code 15/15; Claude Haiku 4.5 · Claude Code 15/15; GPT-6.1 Sol (high) · Codex CLI 15/15; GPT-6.1 Sol (medium) · Codex CLI 15/15; GPT-6.1 Sol (low) · Codex CLI 10/10. Costs are calculations from tokens; medians come from small samples.",
        "sourceIds": [
          "agent-provider-h2h",
          "calc-repricing",
          "price-anthropic",
          "price-openai"
        ]
      }
    ],
    "tables": [
      {
        "id": "h2h-pass-matrix",
        "title": "Pass matrix: configuration × task",
        "columns": [
          {
            "key": "config",
            "label": "Configuration",
            "unit": "text"
          },
          {
            "key": "c0",
            "label": "Fix a buggy median function",
            "unit": "text"
          },
          {
            "key": "c1",
            "label": "Extract invoice fields to JSON",
            "unit": "text"
          },
          {
            "key": "c2",
            "label": "Multi-step shift arithmetic",
            "unit": "text"
          },
          {
            "key": "c3",
            "label": "Refactor recursion to iteration",
            "unit": "text"
          },
          {
            "key": "c4",
            "label": "Classify six support tickets",
            "unit": "text"
          },
          {
            "key": "medianTotal",
            "label": "Median total (s)",
            "unit": "seconds"
          }
        ],
        "rows": [
          {
            "config": "Claude Fable 5.1 · Claude Code",
            "medianTotal": 1.94,
            "c0": "3/3",
            "c1": "3/3",
            "c2": "3/3",
            "c3": "3/3",
            "c4": "3/3"
          },
          {
            "config": "Claude Sonnet 5.5 · Claude Code",
            "medianTotal": 2.31,
            "c0": "3/3",
            "c1": "3/3",
            "c2": "0/3",
            "c3": "3/3",
            "c4": "3/3"
          },
          {
            "config": "Claude Opus 5.5 (high) · Claude Code",
            "medianTotal": 2.71,
            "c0": "3/3",
            "c1": "3/3",
            "c2": "3/3",
            "c3": "3/3",
            "c4": "3/3"
          },
          {
            "config": "Claude Opus 5.5 · Claude Code",
            "medianTotal": 2.75,
            "c0": "3/3",
            "c1": "3/3",
            "c2": "3/3",
            "c3": "3/3",
            "c4": "3/3"
          },
          {
            "config": "Claude Opus 5.5 (low) · Claude Code",
            "medianTotal": 2.83,
            "c0": "3/3",
            "c1": "3/3",
            "c2": "3/3",
            "c3": "3/3",
            "c4": "3/3"
          },
          {
            "config": "Claude Haiku 4.5 · Claude Code",
            "medianTotal": 4.43,
            "c0": "3/3",
            "c1": "3/3",
            "c2": "3/3",
            "c3": "3/3",
            "c4": "3/3"
          },
          {
            "config": "GPT-6.1 Sol (high) · Codex CLI",
            "medianTotal": 5.6,
            "c0": "3/3",
            "c1": "3/3",
            "c2": "3/3",
            "c3": "3/3",
            "c4": "3/3"
          },
          {
            "config": "GPT-6.1 Sol (medium) · Codex CLI",
            "medianTotal": 5.65,
            "c0": "3/3",
            "c1": "3/3",
            "c2": "3/3",
            "c3": "3/3",
            "c4": "3/3"
          },
          {
            "config": "GPT-6.1 Sol (low) · Codex CLI",
            "medianTotal": 6.26,
            "c0": "2/2",
            "c1": "2/2",
            "c2": "2/2",
            "c3": "2/2",
            "c4": "2/2"
          }
        ]
      }
    ],
    "related": [
      "hard-model-head-to-head"
    ]
  },
  "sources": [
    {
      "id": "agent-provider-h2h",
      "title": "Provider head-to-head: Claude Code models vs Codex efforts",
      "kind": "run",
      "date": "2026-10-05",
      "note": "Five short tasks with deterministic validators, declared protocol, every attempt kept.",
      "data": [
        "/benchmarks/raw/provider-h2h/receipts.json"
      ]
    },
    {
      "id": "price-anthropic",
      "title": "Anthropic list prices (Claude models)",
      "kind": "price-list",
      "date": "2026-09-21",
      "url": "https://platform.claude.com/docs/en/about-claude/pricing",
      "note": "Prices as listed by the vendor on 2026-09-21 and recorded in the product price table. Cache reads at the listed rate, one-hour cache writes at twice the input price."
    },
    {
      "id": "price-openai",
      "title": "OpenAI list prices",
      "kind": "price-list",
      "date": "2026-10-03",
      "url": "https://developers.openai.com/api/docs/pricing",
      "note": "Token prices as listed by the vendor on 2026-10-03."
    },
    {
      "id": "calc-repricing",
      "title": "Repricing calculation",
      "kind": "calculation",
      "date": "2026-10-05",
      "note": "Recorded token counts multiplied by the list prices in the price-list sources above. A calculation, not a run: a different model would have used a different number of tokens and reached different outcomes."
    }
  ]
}
