{
  "schema": "agent-public-bench@1",
  "generatedAt": "2026-10-07T00:00:00.000Z",
  "sources": [
    {
      "id": "agent-swebench-c1",
      "title": "Agent on SWE-bench Verified, campaign 1 (25 instances)",
      "kind": "run",
      "date": "2026-10-04",
      "note": "Stratified sample of 25 Verified instances (seed 20261004), one attempt each, official grading harness. Fixed model claude-sonnet-5-5, platform build f0ac3a8a.",
      "data": [
        "/benchmarks/raw/swebench/attempts.json",
        "/benchmarks/raw/swebench/exclusions.json"
      ]
    },
    {
      "id": "agent-swebench-c2",
      "title": "Agent on SWE-bench Verified, campaign 2 (8 compiled-extension instances)",
      "kind": "run",
      "date": "2026-10-05",
      "note": "The 6 compiled-extension instances that campaign 1 could not run, plus 2 replacement candidates. One attempt each, platform build 236c0d3f.",
      "data": [
        "/benchmarks/raw/swebench/attempts.json"
      ]
    },
    {
      "id": "swebench-leaderboard",
      "title": "SWE-bench Verified leaderboard, mini-SWE-agent v2 runs",
      "kind": "public-leaderboard",
      "date": "2026-02-17",
      "url": "https://www.swebench.com",
      "note": "Public per-instance results of 11 models under mini-SWE-agent 2.0.0 (bash only, one attempt). Costs are API list prices as published.",
      "data": [
        "/benchmarks/raw/swebench/panel.json"
      ]
    },
    {
      "id": "swebench-protocol",
      "title": "SWE-bench campaign rules and sample design",
      "kind": "protocol",
      "date": "2026-10-04",
      "note": "Rules declared before the first run: escalations are graded as delivered, gold must resolve on the host, blocked instances are replaced in the same difficulty band, no second attempts. The excluded and replaced instances are listed in the exclusions extract.",
      "data": [
        "/benchmarks/raw/swebench/exclusions.json"
      ]
    },
    {
      "id": "agent-blind-review",
      "title": "Blind review panel: AI worker change vs merged human change",
      "kind": "run",
      "date": "2026-09-28",
      "note": "Each pair is judged by 3 or 4 critic models without labels, in both orders. 5 public OSS tasks and 7 private tasks. Private task names are replaced by neutral labels.",
      "data": [
        "/benchmarks/raw/blind-review/attempts.json"
      ]
    },
    {
      "id": "agent-provider-explorer",
      "title": "Provider explorer receipts: CLI vs API",
      "kind": "run",
      "date": "2026-10-03",
      "note": "230 imported receipts for short fixed tasks over Claude Code CLI, Codex CLI and the OpenAI API, with time to first useful output, total time, tokens and validation.",
      "data": [
        "/benchmarks/raw/provider-explorer/runs.json"
      ]
    },
    {
      "id": "agent-coding-calibration",
      "title": "Coding calibration: fastify/session, h3, uvicorn",
      "kind": "run",
      "date": "2026-10-05",
      "note": "Three real upstream tasks, one attempt per task per platform slice, offline gates against the merged reference.",
      "data": [
        "/benchmarks/raw/calibration/slices.json"
      ]
    },
    {
      "id": "agent-provider-h2h",
      "title": "Provider head-to-head: Claude Code models vs Codex efforts",
      "kind": "run",
      "date": "2026-10-05",
      "note": "Five short tasks with deterministic validators, declared protocol, every attempt kept.",
      "data": [
        "/benchmarks/raw/provider-h2h/receipts.json"
      ]
    },
    {
      "id": "agent-provider-h2h-hard",
      "title": "Provider head-to-head, hard set: eight hard tasks with strict validators",
      "kind": "run",
      "date": "2026-10-06",
      "note": "Eight hard tasks with sandboxed deterministic validators and pre-inference controls, declared protocol, every attempt kept. Format misses are recorded apart from wrong answers.",
      "data": [
        "/benchmarks/raw/provider-h2h-hard/receipts.json"
      ]
    },
    {
      "id": "agent-effort-ladder",
      "title": "Effort ladder: the hard task set at each effort level",
      "kind": "run",
      "date": "2026-10-06",
      "note": "The eight hard tasks of the hard head-to-head at low, medium and high effort (Claude Sonnet 5.5 and Claude Opus 5.5 in Claude Code, GPT-6.1 Sol in Codex CLI), 8 tasks × 2 repetitions per cell, declared protocols, every attempt kept. Efforts the hard head-to-head already ran are reused from its receipts as reference cells.",
      "data": [
        "/benchmarks/raw/effort-ladder/receipts.json",
        "/benchmarks/raw/provider-h2h-hard/receipts.json"
      ]
    },
    {
      "id": "system-one-arena",
      "title": "System One arena: typed-decision models on checkable decisions and in head-to-head games",
      "kind": "run",
      "date": "2026-10-06",
      "note": "Jev 1.13 (TypeSafe API) and six open System One models in llama.cpp 0.6.0 on one Mac Studio (Clef 27B and Clef-Flash 9B Q4_K_M, lev 4B and Kev 4B Q4_K_M, Laya and Julia-1 Q8_0). 1,085 model-blind items in five suites with independent verifiers; exam, latency and repeat passes. Round-robin games (tic-tac-toe, Connect Four, Nim, Dots and Boxes, real-time Pong) against each other and a random and a perfect player, with the featured series and a speed ladder; declared amendments 1 and 2 in the published protocol. Protocol, item hashes and game counts declared before the first counted call; every call and game kept.",
      "data": [
        "/benchmarks/raw/system-one-arena/items.json",
        "/benchmarks/raw/system-one-arena/exam.json",
        "/benchmarks/raw/system-one-arena/latency.json",
        "/benchmarks/raw/system-one-arena/repeat.json",
        "/benchmarks/raw/system-one-arena/same-host-reference.json",
        "/benchmarks/raw/system-one-arena/matches.json",
        "/benchmarks/raw/system-one-arena/match-summary.json",
        "/benchmarks/raw/system-one-arena/featured.json",
        "/benchmarks/raw/system-one-arena/leagues.json",
        "/benchmarks/raw/system-one-arena/speed-lab.json",
        "/benchmarks/raw/system-one-arena/clock-registry.json"
      ]
    },
    {
      "id": "agent-memory-study",
      "title": "Agent memory study: 8 kinds of project memory on Claude Code",
      "kind": "run",
      "date": "2026-10-06",
      "note": "One small Node.js repository, 5 tasks, 8 memory conditions (none, /init, curated, raw notes, dreamed notes, long handbook, Stop hook, curated + hook). Claude Code 2.1.286 headless: Sonnet lane 3 repetitions, Haiku lane 2. Hidden tests and deterministic convention checks; protocol declared before the first session; every attempt kept. The exact memory files and three dreaming passes are published.",
      "data": [
        "/benchmarks/raw/agent-memory/attempts-sonnet.json",
        "/benchmarks/raw/agent-memory/attempts-haiku.json",
        "/benchmarks/raw/agent-memory/dreams.json",
        "/benchmarks/raw/agent-memory/memory-files.json"
      ]
    },
    {
      "id": "agent-coding-agents",
      "title": "Coding agents head-to-head: Claude Code and Codex CLI on 6 hidden-test tasks",
      "kind": "run",
      "date": "2026-10-06",
      "note": "Six small Node.js repositories with hidden tests; controls before the first session (every base fails, every reference passes). Claude Code 2.1.286 with Sonnet 5.5 and Opus 5.5, Codex CLI with GPT-6.1 Sol at medium effort, 2 repetitions per task, OS sandboxes without network, protocol declared before the first session, every session kept. Gemini CLI was probed and not run (browser login).",
      "data": [
        "/benchmarks/raw/coding-agents/sessions.json"
      ]
    },
    {
      "id": "agent-swebench-opus",
      "title": "Agent on SWE-bench Verified with Opus 5.5 as the brain, paired with Sonnet 5.5 (interim)",
      "kind": "run",
      "date": "2026-10-06",
      "note": "8 Verified instances declared before the first run, 2 per difficulty band. One Opus attempt each on platform build 4f6f4027, paired with the earlier Sonnet attempt on the same instance; official grader. Interim: a usage gate stopped the campaign after 3 instances; the other 5 resume after the reset on 2026-10-09.",
      "data": [
        "/benchmarks/raw/swebench-opus/instances.json",
        "/benchmarks/raw/swebench/attempts.json"
      ]
    },
    {
      "id": "agent-caching-consistency",
      "title": "Caching sessions and repeated prompts (Claude Code and Codex CLI)",
      "kind": "run",
      "date": "2026-10-06",
      "note": "Part 1: 5-turn CLI sessions over a fixed synthetic ledger, with the cache counters each provider reports per turn. Part 2: three prompts with deterministic validators, 10 repetitions per model. Declared protocols, validator controls before inference, every attempt kept; answers are published as ordinal ids, never as text.",
      "data": [
        "/benchmarks/raw/caching-consistency/caching.json",
        "/benchmarks/raw/caching-consistency/consistency.json"
      ]
    },
    {
      "id": "agent-routing",
      "title": "Routing runs: Jev router vs LLM routing",
      "kind": "run",
      "date": "2026-10-05",
      "note": "Routing decisions recorded per case and arm.",
      "data": [
        "/benchmarks/raw/routing/receipts.json"
      ]
    },
    {
      "id": "agent-jev-live",
      "title": "Jev live run: 246 timed calls on the 82 routing decisions",
      "kind": "run",
      "date": "2026-10-06",
      "note": "Jev 1.13 called over HTTPS, 3 repeats of the same 82 typed decisions, one call at a time, from one Mac over a home network: client wall time, with the network inside it. The API reports no server time. Cost per 1,000 decisions is a calculation from the reported input tokens and the published price. The case sets were revised against Jev answers, so Jev has a home advantage.",
      "data": [
        "/benchmarks/raw/jev-live/summary.json",
        "/benchmarks/raw/jev-live/calls.json"
      ]
    },
    {
      "id": "price-anthropic",
      "title": "Anthropic list prices (Claude models)",
      "kind": "price-list",
      "date": "2026-09-21",
      "url": "https://platform.claude.com/docs/en/about-claude/pricing",
      "note": "Prices as listed by the vendor on 2026-09-21 and recorded in the product price table. Cache reads at the listed rate, one-hour cache writes at twice the input price."
    },
    {
      "id": "price-google",
      "title": "Google Gemini list prices",
      "kind": "price-list",
      "date": "2026-09-21",
      "url": "https://ai.google.dev/pricing",
      "note": "Gemini 3.x Flash prices as listed by the vendor on 2026-09-21. The vendor announced a doubling from 2027-01-01."
    },
    {
      "id": "price-openai",
      "title": "OpenAI list prices",
      "kind": "price-list",
      "date": "2026-10-03",
      "url": "https://developers.openai.com/api/docs/pricing",
      "note": "Token prices as listed by the vendor on 2026-10-03."
    },
    {
      "id": "price-jev",
      "title": "Jev 1.13 list price",
      "kind": "price-list",
      "date": "2026-09-23",
      "url": "https://docs.typesafe.ai/models",
      "note": "Prices as listed by the vendor on 2026-09-23: input tokens only, output tokens free."
    },
    {
      "id": "agent-routing-overhead",
      "title": "Routing overhead runs: policy microbenchmark and CLI start-up",
      "kind": "run",
      "date": "2026-10-06",
      "note": "In-process timing of the deterministic routing policy (20,000 timed decisions), CLI start-up with a one-word prompt (5 runs per CLI), and decision counts read from recorded bench runs. LLM router timings are reused from the routing runs.",
      "data": [
        "/benchmarks/raw/routing-overhead/results.json"
      ]
    },
    {
      "id": "calc-routing-overhead",
      "title": "Routing overhead per 1,000 tasks (calculation)",
      "kind": "calculation",
      "date": "2026-10-06",
      "note": "Decisions per task from recorded bench runs multiplied by the cost and the median time per decision. Decisions are assumed to wait in line, so the delay is an upper bound. A calculation, not a run.",
      "data": [
        "/benchmarks/raw/routing-overhead/results.json"
      ]
    },
    {
      "id": "openrouter-api-snapshot",
      "title": "OpenRouter public API: models and provider endpoints (snapshot)",
      "kind": "price-list",
      "date": "2026-10-06",
      "url": "https://openrouter.ai/docs/api-reference/list-endpoints-for-a-model",
      "note": "Prices, context, quantization and uptime per provider endpoint as reported by OpenRouter’s public, keyless API on 2026-10-06. Third-party-reported, not measured by Agent. Latency and throughput were not returned.",
      "data": [
        "/benchmarks/raw/provider-index/index.json"
      ]
    },
    {
      "id": "openrouter-fees",
      "title": "OpenRouter pricing and fees",
      "kind": "price-list",
      "date": "2026-10-06",
      "url": "https://openrouter.ai/pricing",
      "note": "OpenRouter states that inference is billed at the provider list price and that its fee is charged when credits are bought (5.5% on Standard by card, $0.80 minimum; 8% on Business; 5% by crypto). Page fetched 2026-10-06.",
      "data": [
        "/benchmarks/raw/provider-index/index.json"
      ]
    },
    {
      "id": "calc-cache-pricing",
      "title": "Cost with and without the prompt cache (calculation)",
      "kind": "calculation",
      "date": "2026-10-06",
      "note": "Recorded tokens per turn × Anthropic list prices. With the cache: uncached input at the input price, cache reads at the cache-read price, 1-hour cache writes at twice the input price, 5-minute writes at 1.25 times (an assumption; none occurred). Without a cache: every input token at the input price. Output is priced the same in both. Not a bill.",
      "data": [
        "/benchmarks/raw/caching-consistency/caching.json"
      ]
    },
    {
      "id": "calc-repricing",
      "title": "Repricing calculation",
      "kind": "calculation",
      "date": "2026-10-05",
      "note": "Recorded token counts multiplied by the list prices in the price-list sources above. A calculation, not a run: a different model would have used a different number of tokens and reached different outcomes."
    },
    {
      "id": "agent-agent-loop",
      "title": "Single call vs agent loop",
      "kind": "run",
      "date": "2026-10-06",
      "data": [
        "/benchmarks/raw/agent-loop/receipts.json"
      ],
      "note": "Receipts of the single call vs agent loop study (public-runs/single-call-vs-agent-loop). Every attempt is kept, failures and contaminated attempts included. Reference single-call cells (Claude Haiku 4.5 and Claude Sonnet 5.5, default effort) are read from raw/provider-h2h-hard."
    },
    {
      "id": "agent-haiku-thinking",
      "title": "Haiku thinking on vs off",
      "kind": "run",
      "date": "2026-10-06",
      "data": [
        "/benchmarks/raw/haiku-thinking/receipts.json"
      ],
      "note": "Receipts of the Haiku thinking study: Claude Haiku 4.5 through Claude Code with thinking off (MAX_THINKING_TOKENS=0) against the recorded thinking-on arms. Routing arms carry per-arm totals computed from each arm’s per-call log and per-case results; hard-task receipts are every attempt of the thinking-off run. Every attempt is kept, failures included. The thinking-on hard-task receipts live in the hard head-to-head extract."
    },
    {
      "id": "agent-structured-output",
      "title": "JSON schema vs instructions",
      "kind": "run",
      "date": "2026-10-07",
      "data": [
        "/benchmarks/raw/structured-output/receipts.json"
      ],
      "note": "Receipts copied from a run of the same three JSON extraction prompts, each asked with instructions only (mode I) and with the CLI’s JSON schema mode (mode S). Prompts, model output and failure reasons are not published; a failed check is named, never quoted. Every attempt is kept. Both routes ran one call at a time. The current protocol file does not verify pre-call registration; see protocolAudit. Calls repeat three fixed hand-made prompts, including a known format-miss case. Both routes used a shared Mac."
    },
    {
      "id": "agent-cache-sessions",
      "title": "Prompt cache across sessions",
      "kind": "run",
      "date": "2026-10-07",
      "data": [
        "/benchmarks/raw/cache-sessions/sessions.json"
      ],
      "note": "Sanitized cache-session receipts: one CLI process per session, 2 turns each, a seeded synthetic ledger (a different seed per condition) and 2 short questions with exact answers. Conditions: A a new temporary working folder per session, B one fixed folder, C a fixed folder with the ledger in the system prompt (Claude Code only). The Codex app-server reports cached input only, so its rows have no cache-write count. Probes are single uncounted calls with their own ledger seed. The answers and the ledger itself are not published. Session 1 was the first use of each setup’s ledger; sessions 2 and 3 reused that ledger. Different seeds prevent full ledger-prefix reuse between setups, but shared CLI-prefix reads remain possible. Correctness uses the reference checker after it trims spaces, surrounding quotes and backticks, a final period and currency units. The surviving protocol file dates from after the counted calls; pre-call declaration is not verified."
    },
    {
      "id": "agent-routing-holdout",
      "title": "Routing on unseen holdout decisions",
      "kind": "run",
      "date": "2026-10-06",
      "data": [
        "/benchmarks/raw/routing-holdout/results.json"
      ],
      "note": "Frozen unseen routing decisions; every router call retained. Costs are calculations from recorded tokens and list prices."
    },
    {
      "id": "agent-speed-anatomy",
      "title": "LLM speed anatomy",
      "kind": "run",
      "date": "2026-10-07",
      "data": [
        "/benchmarks/raw/speed-anatomy/receipts.json"
      ],
      "note": "Receipts of the speed anatomy study: one prompt that asks for 250 numbers in words (output speed, 24 calls) and a seeded synthetic ledger at three sizes with one lookup question (prompt size, 36 calls). Every attempt is kept, failures included. A new seed for every ledger call; no ledger text, prompt or model output is copied."
    },
    {
      "id": "calc-thinking-bill",
      "title": "Reasoning token bill (calculation)",
      "kind": "calculation",
      "date": "2026-10-07",
      "note": "Reported reasoning tokens priced at the recorded list prices. A calculation, not a new run.",
      "data": [
        "/benchmarks/raw/provider-h2h-hard/receipts.json",
        "/benchmarks/raw/effort-ladder/receipts.json",
        "/benchmarks/raw/provider-h2h/receipts.json"
      ]
    },
    {
      "id": "agent-harder-tasks",
      "title": "Harder tasks head-to-head",
      "kind": "run",
      "date": "2026-10-07",
      "data": [
        "/benchmarks/raw/harder-tasks/receipts.json"
      ],
      "note": "Frozen tasks selected with a Sonnet pilot. Fresh counted calls retain failures, format misses and timeouts. Replies and expected answers are not published."
    },
    {
      "id": "calc-latency-budget",
      "title": "Voice-agent latency budget calculation",
      "kind": "calculation",
      "date": "2026-10-07",
      "note": "Recorded decision and first-output times compared with assumed 300, 800 and 1,500 ms budgets. No complete voice turn ran."
    },
    {
      "id": "calc-routing-at-scale",
      "title": "Routing at scale calculation",
      "kind": "calculation",
      "date": "2026-10-07",
      "note": "Recorded routing costs and times scaled to assumed daily volumes. No load test ran; median and p95 scenarios are not measured mean concurrency."
    }
  ],
  "studies": [
    {
      "slug": "swe-bench-verified",
      "title": "Agent on SWE-bench Verified vs 11 public models",
      "seoTitle": "SWE-bench Verified: Agent vs GPT, Claude and Gemini",
      "description": "Agent resolved 25 of 33 SWE-bench Verified instances (76%), inside the public panel's range on the same instances. Cost, time and calls.",
      "question": "How does Agent, a full worker pipeline on one model, do on SWE-bench Verified next to public single-model runs on the very same instances?",
      "answer": "Agent resolved 25 of 33 attempted instances (75.8%, 95% interval 59% to 87%). On the same instances the 11 public mini-SWE-agent v2 runs resolved between 21 and 28 (panel mean 74.1%). Every interval overlaps, so this sample cannot rank Agent above or below any panel model. Agent spent a notional $2.81 and 49 model calls per attempt, with a median of 9.6 minutes; it is slower and more expensive per instance than a bare bash agent because it onboards, plans, verifies and reviews. It resolved 1 of 4 instances that no panel model solved.",
      "date": "2026-10-05",
      "updated": "2026-10-05",
      "tags": [
        "swe-bench",
        "coding-agents",
        "leaderboard",
        "cost",
        "claude-sonnet"
      ],
      "method": [
        "Sample: 25 of the 500 Verified instances, stratified by public difficulty (seed 20261004). Difficulty is how many of the 11 public mini-SWE-agent v2 runs solved the instance.",
        "Campaign 1 ran 25 instances on platform build f0ac3a8a. Six compiled-extension instances could not import in the worker checkout, so the declared rule replaced them in the same band.",
        "Campaign 2 ran those compiled instances (plus two replacement candidates) on build 236c0d3f after a sandbox fix.",
        "One attempt per instance. No retries, no operator answers: an escalation is graded on what was delivered.",
        "Grading uses the official SWE-bench harness and instance images (under amd64 emulation). The gold patch resolved on the host for every instance.",
        "Agent runs its full pipeline (onboarding, research, plan, act, verify, review) on claude-sonnet-5-5 through a subscription CLI. Costs are list-price estimates of the recorded tokens."
      ],
      "caveats": [
        "n = 33: intervals are wide. This is a defect-finding run, not a ranking.",
        "Systems, not models: the panel is one model in a bash-only harness; Agent is a full pipeline on one model.",
        "Panel costs are published API costs; Agent costs are notional subscription estimates, not invoices.",
        "Campaign 1 ended up 19 django, 3 sympy, 2 sphinx and 1 xarray after replacements. Campaign 2 covers the compiled repositories.",
        "Verified issues are public (2015 to 2023) and likely in every model’s training data; contamination is uncontrolled for all systems.",
        "Difficulty bands come from the panel’s own results, so a system outside the panel tends to look better than the panel on hard bands and worse on easy ones (regression to the mean). Read the band chart with that selection effect in mind.",
        "Three campaign-1 empty patches were platform holds before delivery (missing lint tools, a too-literal plan gate, an unanswered question), not wrong fixes. They count as failures here."
      ],
      "sourceIds": [
        "agent-swebench-c1",
        "agent-swebench-c2",
        "swebench-leaderboard",
        "swebench-protocol",
        "price-anthropic"
      ],
      "stats": [
        {
          "id": "agent-rate-33",
          "label": "Agent resolved, all 33 attempted instances",
          "value": 0.7576,
          "unit": "rate",
          "display": "76% (25/33)",
          "n": 33,
          "ci": [
            0.5898,
            0.8717
          ],
          "note": "Both campaigns, one attempt each, failures and empty patches included."
        },
        {
          "id": "agent-rate-c1",
          "label": "Agent resolved, campaign 1 sample of 25",
          "value": 0.72,
          "unit": "rate",
          "display": "72% (18/25)",
          "n": 25,
          "ci": [
            0.5242,
            0.8572
          ]
        },
        {
          "id": "agent-rate-original25",
          "label": "Agent resolved, original seed draw of 25 (no replacements)",
          "value": 0.76,
          "unit": "rate",
          "display": "76% (19/25)",
          "n": 25,
          "ci": [
            0.5657,
            0.885
          ],
          "note": "Mixes two platform builds."
        },
        {
          "id": "panel-mean-33",
          "label": "Public panel mean on the same 33 instances",
          "value": 0.741,
          "unit": "rate",
          "display": "74.1%",
          "n": 33,
          "note": "Mean of 11 public mini-SWE-agent v2 runs."
        },
        {
          "id": "cost-per-attempt",
          "label": "Agent model cost per attempt (notional)",
          "value": 2.81,
          "unit": "usd",
          "display": "$2.81",
          "n": 33,
          "note": "Subscription calls priced at list price; onboarding and failed attempts included."
        },
        {
          "id": "cost-per-resolved",
          "label": "Agent model cost per resolved instance (notional)",
          "value": 3.71,
          "unit": "usd",
          "display": "$3.71",
          "n": 25
        },
        {
          "id": "median-minutes",
          "label": "Median worker time per attempt",
          "value": 9.6,
          "unit": "minutes",
          "display": "9.6 min",
          "n": 33,
          "note": "Range 1.6 to 54.1 min."
        },
        {
          "id": "calls-per-attempt",
          "label": "Model calls per attempt",
          "value": 49.5,
          "unit": "calls",
          "display": "49",
          "n": 33
        }
      ],
      "charts": [
        {
          "id": "swebench-same-instance-leaderboard",
          "title": "Resolved rate on the same 33 SWE-bench Verified instances",
          "subtitle": "Agent vs 11 public mini-SWE-agent v2 runs, one attempt each",
          "kind": "dot-range",
          "unit": "rate",
          "yLabel": "Resolved",
          "series": [
            {
              "name": "Resolved rate",
              "points": [
                {
                  "label": "GPT 5.2 (high)",
                  "value": 0.8485,
                  "lo": 0.6908,
                  "hi": 0.9335,
                  "n": 33
                },
                {
                  "label": "Gemini 3 Flash (high)",
                  "value": 0.8182,
                  "lo": 0.6561,
                  "hi": 0.9139,
                  "n": 33
                },
                {
                  "label": "GLM 5 (high)",
                  "value": 0.7879,
                  "lo": 0.6225,
                  "hi": 0.8932,
                  "n": 33
                },
                {
                  "label": "Agent (Sonnet 5.5, full pipeline)",
                  "value": 0.7576,
                  "lo": 0.5898,
                  "hi": 0.8717,
                  "n": 33,
                  "highlight": true
                },
                {
                  "label": "Claude 4.5 Sonnet (high)",
                  "value": 0.7576,
                  "lo": 0.5898,
                  "hi": 0.8717,
                  "n": 33
                },
                {
                  "label": "Claude 4.5 Haiku (high)",
                  "value": 0.7576,
                  "lo": 0.5898,
                  "hi": 0.8717,
                  "n": 33
                },
                {
                  "label": "Claude 4.5 Opus (high)",
                  "value": 0.7273,
                  "lo": 0.5578,
                  "hi": 0.8493,
                  "n": 33
                },
                {
                  "label": "DeepSeek V3.2 (high)",
                  "value": 0.7273,
                  "lo": 0.5578,
                  "hi": 0.8493,
                  "n": 33
                },
                {
                  "label": "MiniMax M2.5 (high)",
                  "value": 0.697,
                  "lo": 0.5266,
                  "hi": 0.8262,
                  "n": 33
                },
                {
                  "label": "Claude 4.6 Opus",
                  "value": 0.697,
                  "lo": 0.5266,
                  "hi": 0.8262,
                  "n": 33
                },
                {
                  "label": "Kimi K2.5 (high)",
                  "value": 0.697,
                  "lo": 0.5266,
                  "hi": 0.8262,
                  "n": 33
                },
                {
                  "label": "GPT 5 mini",
                  "value": 0.6364,
                  "lo": 0.4662,
                  "hi": 0.7781,
                  "n": 33
                }
              ]
            }
          ],
          "note": "Dots show the rate; whiskers show the 95% Wilson interval for n = 33. The panel compares models under one harness; Agent is a full system on one model.",
          "sourceIds": [
            "agent-swebench-c1",
            "agent-swebench-c2",
            "swebench-leaderboard"
          ]
        },
        {
          "id": "swebench-by-difficulty-band",
          "title": "Resolved rate by difficulty band",
          "subtitle": "Band = how many of the 11 public panel models solved the instance",
          "kind": "grouped-bar",
          "unit": "rate",
          "xLabel": "Difficulty band",
          "yLabel": "Resolved",
          "series": [
            {
              "name": "Agent",
              "points": [
                {
                  "label": "No panel model solved it",
                  "value": 0.25,
                  "lo": 0.0456,
                  "hi": 0.6994,
                  "n": 4,
                  "highlight": true
                },
                {
                  "label": "Under half solved it",
                  "value": 0.75,
                  "lo": 0.3006,
                  "hi": 0.9544,
                  "n": 4,
                  "highlight": true
                },
                {
                  "label": "Half or more solved it",
                  "value": 0.8182,
                  "lo": 0.523,
                  "hi": 0.9486,
                  "n": 11,
                  "highlight": true
                },
                {
                  "label": "Every panel model solved it",
                  "value": 0.8571,
                  "lo": 0.6006,
                  "hi": 0.9599,
                  "n": 14,
                  "highlight": true
                }
              ]
            },
            {
              "name": "Public panel mean",
              "points": [
                {
                  "label": "No panel model solved it",
                  "value": 0,
                  "n": 4
                },
                {
                  "label": "Under half solved it",
                  "value": 0.2727,
                  "n": 4
                },
                {
                  "label": "Half or more solved it",
                  "value": 0.8512,
                  "n": 11
                },
                {
                  "label": "Every panel model solved it",
                  "value": 1,
                  "n": 14
                }
              ]
            }
          ],
          "note": "Agent whiskers are 95% Wilson intervals. The \"no panel model solved it\" band has 4 instances; Agent resolved 1 (matplotlib__matplotlib-21568).",
          "sourceIds": [
            "agent-swebench-c1",
            "agent-swebench-c2",
            "swebench-leaderboard",
            "swebench-protocol"
          ]
        },
        {
          "id": "swebench-cost-vs-resolved",
          "title": "Cost per instance vs resolved rate",
          "subtitle": "Same 33 instances. Panel = API list price; Agent = notional subscription estimate",
          "kind": "scatter",
          "unit": "rate",
          "xLabel": "Mean model cost per instance (USD)",
          "yLabel": "Resolved rate",
          "series": [
            {
              "name": "Public panel (mini-SWE-agent v2)",
              "points": [
                {
                  "label": "Claude 4.5 Opus (high)",
                  "x": 0.861,
                  "value": 0.7273,
                  "n": 33
                },
                {
                  "label": "Gemini 3 Flash (high)",
                  "x": 0.357,
                  "value": 0.8182,
                  "n": 33
                },
                {
                  "label": "MiniMax M2.5 (high)",
                  "x": 0.075,
                  "value": 0.697,
                  "n": 33
                },
                {
                  "label": "Claude 4.6 Opus",
                  "x": 0.61,
                  "value": 0.697,
                  "n": 33
                },
                {
                  "label": "GLM 5 (high)",
                  "x": 0.525,
                  "value": 0.7879,
                  "n": 33
                },
                {
                  "label": "GPT 5.2 (high)",
                  "x": 0.533,
                  "value": 0.8485,
                  "n": 33
                },
                {
                  "label": "Claude 4.5 Sonnet (high)",
                  "x": 0.692,
                  "value": 0.7576,
                  "n": 33
                },
                {
                  "label": "Kimi K2.5 (high)",
                  "x": 0.179,
                  "value": 0.697,
                  "n": 33
                },
                {
                  "label": "DeepSeek V3.2 (high)",
                  "x": 0.463,
                  "value": 0.7273,
                  "n": 33
                },
                {
                  "label": "Claude 4.5 Haiku (high)",
                  "x": 0.363,
                  "value": 0.7576,
                  "n": 33
                },
                {
                  "label": "GPT 5 mini",
                  "x": 0.051,
                  "value": 0.6364,
                  "n": 33
                }
              ]
            },
            {
              "name": "Agent",
              "points": [
                {
                  "label": "Agent",
                  "x": 2.807,
                  "value": 0.7576,
                  "n": 33,
                  "highlight": true
                }
              ]
            }
          ],
          "note": "Agent's cost includes repository onboarding, planning, verification and review; it is a list-price estimate for subscription calls, not an invoice. Panel costs are published API costs.",
          "sourceIds": [
            "agent-swebench-c1",
            "agent-swebench-c2",
            "swebench-leaderboard",
            "price-anthropic"
          ]
        },
        {
          "id": "swebench-model-calls",
          "title": "Model calls per instance",
          "subtitle": "Mean over the same 33 instances",
          "kind": "bar",
          "unit": "calls",
          "yLabel": "Calls per instance",
          "series": [
            {
              "name": "Mean calls",
              "points": [
                {
                  "label": "DeepSeek V3.2 (high)",
                  "value": 88.2,
                  "n": 33
                },
                {
                  "label": "GLM 5 (high)",
                  "value": 77.5,
                  "n": 33
                },
                {
                  "label": "Claude 4.5 Haiku (high)",
                  "value": 68.5,
                  "n": 33
                },
                {
                  "label": "MiniMax M2.5 (high)",
                  "value": 58.4,
                  "n": 33
                },
                {
                  "label": "Kimi K2.5 (high)",
                  "value": 56.7,
                  "n": 33
                },
                {
                  "label": "Gemini 3 Flash (high)",
                  "value": 54.2,
                  "n": 33
                },
                {
                  "label": "Claude 4.5 Sonnet (high)",
                  "value": 51,
                  "n": 33
                },
                {
                  "label": "Agent",
                  "value": 49.5,
                  "n": 33,
                  "highlight": true
                },
                {
                  "label": "Claude 4.5 Opus (high)",
                  "value": 35.9,
                  "n": 33
                },
                {
                  "label": "GPT 5.2 (high)",
                  "value": 35.6,
                  "n": 33
                },
                {
                  "label": "Claude 4.6 Opus",
                  "value": 28.9,
                  "n": 33
                },
                {
                  "label": "GPT 5 mini",
                  "value": 20.8,
                  "n": 33
                }
              ]
            }
          ],
          "note": "A panel call is one bash-agent step. An Agent call is one model request of any stage (research, plan, act, verify, review).",
          "sourceIds": [
            "agent-swebench-c1",
            "agent-swebench-c2",
            "swebench-leaderboard"
          ]
        },
        {
          "id": "swebench-cost-by-stage",
          "title": "Where Agent's model spend goes",
          "subtitle": "Share of notional model cost by stage, all 33 attempts",
          "kind": "bar",
          "unit": "usd",
          "yLabel": "USD (notional)",
          "series": [
            {
              "name": "Cost",
              "points": [
                {
                  "label": "Act (edit and run)",
                  "value": 44.05
                },
                {
                  "label": "Research",
                  "value": 20.56
                },
                {
                  "label": "Verify",
                  "value": 8.72
                },
                {
                  "label": "Other",
                  "value": 6.24
                },
                {
                  "label": "Review",
                  "value": 5.58
                },
                {
                  "label": "Context compaction",
                  "value": 5.41
                },
                {
                  "label": "Onboarding notes",
                  "value": 2.09
                }
              ]
            }
          ],
          "note": "Total $92.64 over 33 attempts.",
          "sourceIds": [
            "agent-swebench-c1",
            "agent-swebench-c2"
          ]
        },
        {
          "id": "swebench-views",
          "title": "Every way to slice the run, with intervals",
          "subtitle": "Agent resolved rate and 95% Wilson interval per declared view",
          "kind": "dot-range",
          "unit": "rate",
          "yLabel": "Resolved",
          "series": [
            {
              "name": "Agent",
              "points": [
                {
                  "label": "Campaign 1: 25-instance sample",
                  "value": 0.72,
                  "lo": 0.5242,
                  "hi": 0.8572,
                  "n": 25
                },
                {
                  "label": "Campaign 2: 8 compiled-extension instances",
                  "value": 0.875,
                  "lo": 0.5291,
                  "hi": 0.9776,
                  "n": 8
                },
                {
                  "label": "Original seed draw of 25",
                  "value": 0.76,
                  "lo": 0.5657,
                  "hi": 0.885,
                  "n": 25
                },
                {
                  "label": "All 33 attempted",
                  "value": 0.7576,
                  "lo": 0.5898,
                  "hi": 0.8717,
                  "n": 33,
                  "highlight": true
                }
              ]
            },
            {
              "name": "Public panel mean, same instances",
              "points": [
                {
                  "label": "Campaign 1: 25-instance sample",
                  "value": 0.7091,
                  "n": 25
                },
                {
                  "label": "Campaign 2: 8 compiled-extension instances",
                  "value": 0.8409,
                  "n": 8
                },
                {
                  "label": "Original seed draw of 25",
                  "value": 0.7091,
                  "n": 25
                },
                {
                  "label": "All 33 attempted",
                  "value": 0.741,
                  "n": 33
                }
              ]
            }
          ],
          "note": "The original draw and \"all 33\" mix two platform builds.",
          "sourceIds": [
            "agent-swebench-c1",
            "agent-swebench-c2",
            "swebench-leaderboard",
            "swebench-protocol"
          ]
        }
      ],
      "tables": [
        {
          "id": "swebench-per-instance",
          "title": "Every attempt",
          "columns": [
            {
              "key": "instance",
              "label": "Instance",
              "unit": "text"
            },
            {
              "key": "band",
              "label": "Panel solved",
              "unit": "text"
            },
            {
              "key": "outcome",
              "label": "Agent outcome",
              "unit": "text"
            },
            {
              "key": "minutes",
              "label": "Minutes",
              "unit": "minutes"
            },
            {
              "key": "calls",
              "label": "Calls",
              "unit": "calls"
            },
            {
              "key": "costUsd",
              "label": "Cost (notional)",
              "unit": "usd"
            },
            {
              "key": "campaign",
              "label": "Campaign",
              "unit": "text"
            }
          ],
          "rows": [
            {
              "instance": "astropy__astropy-13579",
              "band": "11/11",
              "outcome": "resolved",
              "minutes": 54.1,
              "calls": 65,
              "costUsd": 4,
              "campaign": "2"
            },
            {
              "instance": "astropy__astropy-14096",
              "band": "10/11",
              "outcome": "resolved",
              "minutes": 10.4,
              "calls": 64,
              "costUsd": 3.35,
              "campaign": "2"
            },
            {
              "instance": "astropy__astropy-7336",
              "band": "11/11",
              "outcome": "unresolved",
              "minutes": 23,
              "calls": 48,
              "costUsd": 2.03,
              "campaign": "2"
            },
            {
              "instance": "django__django-10554",
              "band": "0/11",
              "outcome": "unresolved",
              "minutes": 4.7,
              "calls": 34,
              "costUsd": 1.52,
              "campaign": "1"
            },
            {
              "instance": "django__django-11333",
              "band": "11/11",
              "outcome": "resolved",
              "minutes": 8,
              "calls": 46,
              "costUsd": 2.51,
              "campaign": "1"
            },
            {
              "instance": "django__django-11885",
              "band": "3/11",
              "outcome": "resolved",
              "minutes": 7.3,
              "calls": 38,
              "costUsd": 2.27,
              "campaign": "1"
            },
            {
              "instance": "django__django-12193",
              "band": "10/11",
              "outcome": "resolved",
              "minutes": 5.7,
              "calls": 43,
              "costUsd": 2.06,
              "campaign": "1"
            },
            {
              "instance": "django__django-12741",
              "band": "11/11",
              "outcome": "resolved",
              "minutes": 10.5,
              "calls": 54,
              "costUsd": 2.98,
              "campaign": "1"
            },
            {
              "instance": "django__django-13158",
              "band": "10/11",
              "outcome": "resolved",
              "minutes": 7,
              "calls": 43,
              "costUsd": 2.15,
              "campaign": "1"
            },
            {
              "instance": "django__django-13925",
              "band": "8/11",
              "outcome": "resolved",
              "minutes": 6.8,
              "calls": 43,
              "costUsd": 2.29,
              "campaign": "1"
            },
            {
              "instance": "django__django-14034",
              "band": "0/11",
              "outcome": "unresolved",
              "minutes": 11.5,
              "calls": 54,
              "costUsd": 3.03,
              "campaign": "1"
            },
            {
              "instance": "django__django-14373",
              "band": "11/11",
              "outcome": "resolved",
              "minutes": 5.7,
              "calls": 38,
              "costUsd": 1.74,
              "campaign": "1"
            },
            {
              "instance": "django__django-14559",
              "band": "11/11",
              "outcome": "resolved",
              "minutes": 7.2,
              "calls": 47,
              "costUsd": 2.71,
              "campaign": "1"
            },
            {
              "instance": "django__django-14631",
              "band": "9/11",
              "outcome": "empty patch",
              "minutes": 38.2,
              "calls": 47,
              "costUsd": 2.8,
              "campaign": "1"
            },
            {
              "instance": "django__django-14672",
              "band": "11/11",
              "outcome": "resolved",
              "minutes": 9.1,
              "calls": 58,
              "costUsd": 2.99,
              "campaign": "1"
            },
            {
              "instance": "django__django-15022",
              "band": "4/11",
              "outcome": "unresolved",
              "minutes": 15,
              "calls": 62,
              "costUsd": 4.1,
              "campaign": "1"
            },
            {
              "instance": "django__django-15280",
              "band": "7/11",
              "outcome": "resolved",
              "minutes": 11,
              "calls": 59,
              "costUsd": 3.67,
              "campaign": "1"
            },
            {
              "instance": "django__django-15572",
              "band": "10/11",
              "outcome": "resolved",
              "minutes": 5.3,
              "calls": 42,
              "costUsd": 1.81,
              "campaign": "1"
            },
            {
              "instance": "django__django-15732",
              "band": "3/11",
              "outcome": "resolved",
              "minutes": 10.9,
              "calls": 59,
              "costUsd": 2.9,
              "campaign": "1"
            },
            {
              "instance": "django__django-16255",
              "band": "10/11",
              "outcome": "resolved",
              "minutes": 7.6,
              "calls": 47,
              "costUsd": 2.55,
              "campaign": "1"
            },
            {
              "instance": "django__django-16333",
              "band": "11/11",
              "outcome": "resolved",
              "minutes": 13,
              "calls": 37,
              "costUsd": 1.61,
              "campaign": "1"
            },
            {
              "instance": "django__django-17029",
              "band": "11/11",
              "outcome": "resolved",
              "minutes": 7.8,
              "calls": 42,
              "costUsd": 2.28,
              "campaign": "1"
            },
            {
              "instance": "matplotlib__matplotlib-21568",
              "band": "0/11",
              "outcome": "resolved",
              "minutes": 9.4,
              "calls": 53,
              "costUsd": 3.02,
              "campaign": "2"
            },
            {
              "instance": "matplotlib__matplotlib-24149",
              "band": "10/11",
              "outcome": "resolved",
              "minutes": 41.1,
              "calls": 64,
              "costUsd": 3.64,
              "campaign": "2"
            },
            {
              "instance": "matplotlib__matplotlib-24970",
              "band": "11/11",
              "outcome": "resolved",
              "minutes": 33,
              "calls": 66,
              "costUsd": 3.9,
              "campaign": "2"
            },
            {
              "instance": "pydata__xarray-6721",
              "band": "11/11",
              "outcome": "resolved",
              "minutes": 16.2,
              "calls": 57,
              "costUsd": 4.09,
              "campaign": "1"
            },
            {
              "instance": "scikit-learn__scikit-learn-14894",
              "band": "11/11",
              "outcome": "resolved",
              "minutes": 35.2,
              "calls": 50,
              "costUsd": 3.23,
              "campaign": "2"
            },
            {
              "instance": "scikit-learn__scikit-learn-25973",
              "band": "10/11",
              "outcome": "resolved",
              "minutes": 8.6,
              "calls": 44,
              "costUsd": 2.43,
              "campaign": "2"
            },
            {
              "instance": "sphinx-doc__sphinx-10435",
              "band": "2/11",
              "outcome": "resolved",
              "minutes": 16.8,
              "calls": 67,
              "costUsd": 4.83,
              "campaign": "1"
            },
            {
              "instance": "sphinx-doc__sphinx-9698",
              "band": "11/11",
              "outcome": "resolved",
              "minutes": 9.6,
              "calls": 49,
              "costUsd": 2.69,
              "campaign": "1"
            },
            {
              "instance": "sympy__sympy-18189",
              "band": "11/11",
              "outcome": "empty patch",
              "minutes": 1.6,
              "calls": 13,
              "costUsd": 0.89,
              "campaign": "1"
            },
            {
              "instance": "sympy__sympy-20428",
              "band": "0/11",
              "outcome": "unresolved",
              "minutes": 21,
              "calls": 71,
              "costUsd": 4.81,
              "campaign": "1"
            },
            {
              "instance": "sympy__sympy-22456",
              "band": "9/11",
              "outcome": "empty patch",
              "minutes": 6.4,
              "calls": 28,
              "costUsd": 1.75,
              "campaign": "1"
            }
          ]
        },
        {
          "id": "swebench-panel",
          "title": "Public panel on the same instances",
          "columns": [
            {
              "key": "system",
              "label": "System",
              "unit": "text"
            },
            {
              "key": "resolved",
              "label": "Resolved of 33",
              "unit": "count"
            },
            {
              "key": "rate",
              "label": "Rate",
              "unit": "rate"
            },
            {
              "key": "board500",
              "label": "Full Verified board (500)",
              "unit": "percent"
            },
            {
              "key": "meanCost",
              "label": "Mean cost per instance, these 33",
              "unit": "usd"
            }
          ],
          "rows": [
            {
              "system": "GPT 5.2 (high)",
              "resolved": 28,
              "rate": 0.8485,
              "board500": 72.8,
              "meanCost": 0.533
            },
            {
              "system": "Gemini 3 Flash (high)",
              "resolved": 27,
              "rate": 0.8182,
              "board500": 75.8,
              "meanCost": 0.357
            },
            {
              "system": "GLM 5 (high)",
              "resolved": 26,
              "rate": 0.7879,
              "board500": 72.8,
              "meanCost": 0.525
            },
            {
              "system": "Agent (Sonnet 5.5, full pipeline)",
              "resolved": 25,
              "rate": 0.7576,
              "board500": null,
              "meanCost": 2.81
            },
            {
              "system": "Claude 4.5 Sonnet (high)",
              "resolved": 25,
              "rate": 0.7576,
              "board500": 71.4,
              "meanCost": 0.692
            },
            {
              "system": "Claude 4.5 Haiku (high)",
              "resolved": 25,
              "rate": 0.7576,
              "board500": 66.6,
              "meanCost": 0.363
            },
            {
              "system": "Claude 4.5 Opus (high)",
              "resolved": 24,
              "rate": 0.7273,
              "board500": 76.8,
              "meanCost": 0.861
            },
            {
              "system": "DeepSeek V3.2 (high)",
              "resolved": 24,
              "rate": 0.7273,
              "board500": 70,
              "meanCost": 0.463
            },
            {
              "system": "MiniMax M2.5 (high)",
              "resolved": 23,
              "rate": 0.697,
              "board500": 75.8,
              "meanCost": 0.075
            },
            {
              "system": "Claude 4.6 Opus",
              "resolved": 23,
              "rate": 0.697,
              "board500": 75.6,
              "meanCost": 0.61
            },
            {
              "system": "Kimi K2.5 (high)",
              "resolved": 23,
              "rate": 0.697,
              "board500": 70.8,
              "meanCost": 0.179
            },
            {
              "system": "GPT 5 mini",
              "resolved": 21,
              "rate": 0.6364,
              "board500": 56.2,
              "meanCost": 0.051
            }
          ]
        }
      ]
    },
    {
      "slug": "swe-bench-opus-vs-sonnet",
      "title": "Opus 5.5 vs Sonnet 5.5 as the Agent brain on SWE-bench Verified (interim)",
      "seoTitle": "Opus 5.5 vs Sonnet 5.5 on SWE-bench Verified (interim)",
      "description": "Interim: 3 of 8 paired SWE-bench Verified instances graded. Opus 5.5 resolved 2, Sonnet 5.5 1 (McNemar p = 1.0). Cost 2.6×, a list-price calculation.",
      "question": "Does Agent resolve more SWE-bench Verified instances with Claude Opus 5.5 as its brain than with Claude Sonnet 5.5, and at what cost and time?",
      "answer": "Interim, not a final result: 3 of 8 declared pairs are graded. With Claude Opus 5.5 as its brain, Agent resolved 2 of 3 (95% Wilson 21–94%); with Claude Sonnet 5.5 it resolved 1 of 3 (6–79%) on the same instances. The one discordant instance (django__django-10554) went to Opus, and the exact McNemar p is 1.0, so these pairs show no difference. On the same instances Opus cost 2.6× as much, a list-price calculation ($22.76 vs $8.64), and used 1.9× the worker minutes. The arms ran on different platform builds, and the other 5 pairs remain not started in this dataset; the recorded usage-reset date was 2026-10-09.",
      "date": "2026-10-06",
      "updated": "2026-10-06",
      "tags": [
        "swe-bench",
        "claude-opus",
        "claude-sonnet",
        "agent-harness",
        "interim"
      ],
      "method": [
        "A paired probe declared before any Opus run: 8 SWE-bench Verified instances from the 33 that Agent attempted with Sonnet, 2 per difficulty band (1 Sonnet miss and 1 Sonnet resolve each), picked by a fixed rule.",
        "Opus arm: Agent with every model call on Claude Opus 5.5 (routing off), balanced mode, real onboarding, cold start, no cost or time cap, one attempt per instance, on platform build 4f6f4027. Sonnet arm: the earlier campaign attempt on the same instance with the same flags and task spec (builds f0ac3a8a and 236c0d3f).",
        "Grading: Official SWE-bench harness 5.0.2 with the official instance images, amd64 emulation; the gold patch must resolve on this host first. An escalation is graded as delivered; nobody answers it.",
        "A usage gate starts an instance only when the subscription has 3% or more left in both its weekly and 5-hour windows. The run ledger resumes after a halt and never repeats an attempt.",
        "Notional cost: the platform price table at each run commit applied to the recorded tokens. The exact McNemar test reads only the discordant pairs."
      ],
      "caveats": [
        "Interim: 3 of 8 declared pairs. n = 3 supports no ranking: the 95% Wilson intervals overlap almost completely.",
        "Different platform builds: Opus ran on 4f6f4027; the Sonnet attempts ran on f0ac3a8a and 236c0d3f. Platform changes can move results by themselves, so a gap compares Sonnet on an older platform with Opus on the current one, not the models alone.",
        "2 of the Sonnet misses in the declared set (django__django-14631, sympy__sympy-18189) were empty patches from platform holds, not wrong fixes. They are not graded for Opus yet.",
        "The instances were chosen to hold Sonnet misses and resolves in equal numbers, so neither rate estimates SWE-bench Verified as a whole.",
        "Costs are list-price calculations on subscription runs, not invoices. Grading ran under amd64 emulation; contamination is not controlled."
      ],
      "sourceIds": [
        "agent-swebench-opus",
        "agent-swebench-c1",
        "agent-swebench-c2",
        "calc-repricing",
        "price-anthropic"
      ],
      "stats": [
        {
          "id": "swebench-opus-resolved",
          "label": "Claude Opus 5.5 in Agent: resolved, interim (3 of 8 pairs graded)",
          "value": 0.6667,
          "unit": "rate",
          "display": "67% (2/3)",
          "n": 3,
          "ci": [
            0.2077,
            0.9385
          ]
        },
        {
          "id": "swebench-sonnet-resolved",
          "label": "Claude Sonnet 5.5 in Agent: resolved on the same 3 instances",
          "value": 0.3333,
          "unit": "rate",
          "display": "33% (1/3)",
          "n": 3,
          "ci": [
            0.0615,
            0.7923
          ]
        },
        {
          "id": "swebench-opus-sonnet-mcnemar",
          "label": "Exact McNemar p on the graded pairs",
          "value": 1,
          "unit": "score",
          "display": "p = 1.0",
          "n": 3,
          "note": "1 pair where only Opus resolved, 0 pairs where only Sonnet did. The test reads only these; p = 1.0 is no evidence of a difference."
        },
        {
          "id": "swebench-opus-sonnet-graded",
          "label": "Declared pairs graded so far",
          "value": 3,
          "unit": "count",
          "display": "3 of 8",
          "n": 8,
          "note": "5 not started: the usage gate stopped the campaign before django__django-11885. The recorded usage-reset date was 2026-10-09; no resumed attempts are included here. Every started attempt counts."
        },
        {
          "id": "swebench-opus-sonnet-cost-ratio",
          "label": "Opus vs Sonnet list-price cost on the same instances (calculation)",
          "value": 2.63,
          "unit": "ratio",
          "display": "2.6×",
          "n": 3,
          "note": "$22.76 vs $8.64 over the 3 graded instances: notional cost from the platform price table, not an invoice."
        },
        {
          "id": "swebench-opus-sonnet-minutes-ratio",
          "label": "Opus vs Sonnet worker minutes on the same instances (ratio of totals)",
          "value": 1.9,
          "unit": "ratio",
          "display": "1.9×",
          "n": 3,
          "note": "55.3 vs 29.1 worker minutes."
        }
      ],
      "charts": [
        {
          "id": "swebench-opus-sonnet-resolved",
          "title": "Resolved on the same 3 SWE-bench Verified instances (interim)",
          "subtitle": "One attempt per arm per instance, official grader · 95% Wilson intervals",
          "kind": "dot-range",
          "unit": "rate",
          "whisker": "ci95",
          "yLabel": "Resolved",
          "viz": "IntervalDotPlot",
          "series": [
            {
              "name": "Resolved",
              "points": [
                {
                  "label": "Claude Opus 5.5 (Agent, new build)",
                  "value": 0.6667,
                  "lo": 0.2077,
                  "hi": 0.9385,
                  "n": 3
                },
                {
                  "label": "Claude Sonnet 5.5 (Agent, older builds)",
                  "value": 0.3333,
                  "lo": 0.0615,
                  "hi": 0.7923,
                  "n": 3
                }
              ]
            }
          ],
          "note": "Interim: 3 of 8 declared pairs are graded; 5 were never started; no resumed attempts are included here. With n = 3 the intervals span most of the axis, so this chart supports no ranking. The instances were chosen to hold Sonnet misses and resolves in equal numbers, so neither rate estimates SWE-bench Verified as a whole.",
          "sourceIds": [
            "agent-swebench-opus",
            "agent-swebench-c1",
            "agent-swebench-c2"
          ]
        },
        {
          "id": "swebench-opus-sonnet-cost-per-attempt",
          "title": "List-price cost per attempt (calculation)",
          "subtitle": "Mean over the same 3 instances; notional cost from the platform price table",
          "kind": "bar",
          "unit": "usd",
          "yLabel": "USD per attempt",
          "viz": "CostBars",
          "series": [
            {
              "name": "List-price cost per attempt",
              "points": [
                {
                  "label": "Claude Opus 5.5 (Agent, new build)",
                  "value": 7.59,
                  "n": 3
                },
                {
                  "label": "Claude Sonnet 5.5 (Agent, older builds)",
                  "value": 2.88,
                  "n": 3
                }
              ]
            }
          ],
          "note": "Calculation, not an invoice: the platform price table at each run commit applied to the recorded tokens of subscription runs (Opus 5.5: $4 input, $20 output, $0.20 cache read per million tokens). Totals $22.76 vs $8.64: 2.6×, a ratio of two calculations.",
          "sourceIds": [
            "agent-swebench-opus",
            "agent-swebench-c1",
            "agent-swebench-c2",
            "calc-repricing",
            "price-anthropic"
          ]
        },
        {
          "id": "swebench-opus-sonnet-cost-by-instance",
          "title": "List-price cost per instance (calculation)",
          "subtitle": "One attempt per arm; notional cost from the platform price table",
          "kind": "grouped-bar",
          "unit": "usd",
          "xLabel": "Instance",
          "yLabel": "USD",
          "viz": "DumbbellPairs",
          "series": [
            {
              "name": "Claude Sonnet 5.5 (Agent, older builds)",
              "points": [
                {
                  "label": "django-10554",
                  "value": 1.52,
                  "n": 1
                },
                {
                  "label": "matplotlib-21568",
                  "value": 3.02,
                  "n": 1
                },
                {
                  "label": "django-15022",
                  "value": 4.1,
                  "n": 1
                }
              ]
            },
            {
              "name": "Claude Opus 5.5 (Agent, new build)",
              "points": [
                {
                  "label": "django-10554",
                  "value": 9.99,
                  "n": 1
                },
                {
                  "label": "matplotlib-21568",
                  "value": 4.9,
                  "n": 1
                },
                {
                  "label": "django-15022",
                  "value": 7.87,
                  "n": 1
                }
              ]
            }
          ],
          "note": "Calculation, not an invoice. Outcomes: django-10554 Opus resolved, Sonnet unresolved; matplotlib-21568 Opus resolved, Sonnet resolved; django-15022 Opus unresolved, Sonnet unresolved.",
          "sourceIds": [
            "agent-swebench-opus",
            "agent-swebench-c1",
            "agent-swebench-c2",
            "calc-repricing",
            "price-anthropic"
          ]
        },
        {
          "id": "swebench-opus-sonnet-minutes",
          "title": "Worker time per attempt",
          "subtitle": "Median minutes; whiskers = fastest and slowest of 3 attempts (not an interval)",
          "kind": "dot-range",
          "unit": "minutes",
          "whisker": "minmax",
          "yLabel": "Minutes",
          "viz": "LatencyLanes",
          "series": [
            {
              "name": "Worker minutes per attempt",
              "points": [
                {
                  "label": "Claude Opus 5.5 (Agent, new build)",
                  "value": 20.26,
                  "lo": 9.76,
                  "hi": 25.29,
                  "n": 3
                },
                {
                  "label": "Claude Sonnet 5.5 (Agent, older builds)",
                  "value": 9.37,
                  "lo": 4.74,
                  "hi": 15,
                  "n": 3
                }
              ]
            }
          ],
          "note": "Worker minutes from the run reports. Opus attempts ran one at a time on the newer build; the Sonnet attempts ran in earlier campaigns. 3 attempts per arm is too few to call a difference; a range is not a confidence interval.",
          "sourceIds": [
            "agent-swebench-opus",
            "agent-swebench-c1",
            "agent-swebench-c2"
          ]
        },
        {
          "id": "swebench-opus-sonnet-stage-cost",
          "title": "Where the cost goes: pipeline stages (calculation)",
          "subtitle": "List-price cost per stage, summed over the same 3 instances",
          "kind": "grouped-bar",
          "unit": "usd",
          "xLabel": "Pipeline stage",
          "yLabel": "USD",
          "viz": "DumbbellPairs",
          "series": [
            {
              "name": "Claude Sonnet 5.5 (Agent, older builds)",
              "points": [
                {
                  "label": "Research",
                  "value": 2.34,
                  "n": 3
                },
                {
                  "label": "Plan",
                  "value": 1.16,
                  "n": 3
                },
                {
                  "label": "Implement",
                  "value": 1.32,
                  "n": 3
                },
                {
                  "label": "Validate",
                  "value": 1.42,
                  "n": 3
                },
                {
                  "label": "Review",
                  "value": 0.51,
                  "n": 3
                },
                {
                  "label": "Deliver",
                  "value": 1.56,
                  "n": 3
                }
              ]
            },
            {
              "name": "Claude Opus 5.5 (Agent, new build)",
              "points": [
                {
                  "label": "Research",
                  "value": 6.28,
                  "n": 3
                },
                {
                  "label": "Plan",
                  "value": 1.11,
                  "n": 3
                },
                {
                  "label": "Implement",
                  "value": 4.94,
                  "n": 3
                },
                {
                  "label": "Validate",
                  "value": 2.97,
                  "n": 3
                },
                {
                  "label": "Review",
                  "value": 2.01,
                  "n": 3
                },
                {
                  "label": "Deliver",
                  "value": 4.34,
                  "n": 3
                }
              ]
            }
          ],
          "note": "Calculation, not an invoice: stage costs from the run reports (run-mode stages). Onboarding and calls outside a stage are not in these bars, so the stages sum to less than the attempt totals.",
          "sourceIds": [
            "agent-swebench-opus",
            "agent-swebench-c1",
            "agent-swebench-c2",
            "calc-repricing",
            "price-anthropic"
          ]
        }
      ],
      "tables": [
        {
          "id": "swebench-opus-sonnet-pairs",
          "title": "Same 3 instances, two brains",
          "viz": "PairedOutcomeGrid",
          "columns": [
            {
              "key": "pair",
              "label": "Pair",
              "unit": "text"
            },
            {
              "key": "bothRight",
              "label": "Both resolved",
              "unit": "count"
            },
            {
              "key": "onlyA",
              "label": "Only Opus resolved",
              "unit": "count"
            },
            {
              "key": "onlyB",
              "label": "Only Sonnet resolved",
              "unit": "count"
            },
            {
              "key": "bothWrong",
              "label": "Neither resolved",
              "unit": "count"
            },
            {
              "key": "p",
              "label": "Exact McNemar p"
            }
          ],
          "rows": [
            {
              "pair": "Claude Opus 5.5 vs Claude Sonnet 5.5",
              "bothRight": 1,
              "onlyA": 1,
              "onlyB": 0,
              "bothWrong": 1,
              "p": 1
            }
          ]
        },
        {
          "id": "swebench-opus-sonnet-graded",
          "title": "Graded instances: resolved or not, per brain",
          "viz": "HeatMatrix",
          "columns": [
            {
              "key": "instance",
              "label": "Instance",
              "unit": "text"
            },
            {
              "key": "band",
              "label": "Difficulty band",
              "unit": "text"
            },
            {
              "key": "opus",
              "label": "Claude Opus 5.5 (Agent, new build)",
              "unit": "text"
            },
            {
              "key": "sonnet",
              "label": "Claude Sonnet 5.5 (Agent, older builds)",
              "unit": "text"
            }
          ],
          "rows": [
            {
              "instance": "django__django-10554",
              "band": "No panel model solved it",
              "opus": "resolved",
              "sonnet": "unresolved"
            },
            {
              "instance": "matplotlib__matplotlib-21568",
              "band": "No panel model solved it",
              "opus": "resolved",
              "sonnet": "resolved"
            },
            {
              "instance": "django__django-15022",
              "band": "Under half solved it",
              "opus": "unresolved",
              "sonnet": "unresolved"
            }
          ]
        },
        {
          "id": "swebench-opus-sonnet-instances",
          "title": "All 8 declared instances, in run order (an escalation is graded as delivered)",
          "columns": [
            {
              "key": "instance",
              "label": "Instance",
              "unit": "text"
            },
            {
              "key": "band",
              "label": "Difficulty band",
              "unit": "text"
            },
            {
              "key": "panel",
              "label": "Panel solved (of 11)",
              "unit": "count"
            },
            {
              "key": "opus",
              "label": "Opus 5.5",
              "unit": "text"
            },
            {
              "key": "sonnet",
              "label": "Sonnet 5.5",
              "unit": "text"
            },
            {
              "key": "opusCost",
              "label": "Opus cost (calc.)",
              "unit": "usd"
            },
            {
              "key": "sonnetCost",
              "label": "Sonnet cost (calc.)",
              "unit": "usd"
            },
            {
              "key": "opusMin",
              "label": "Opus min",
              "unit": "minutes"
            },
            {
              "key": "sonnetMin",
              "label": "Sonnet min",
              "unit": "minutes"
            }
          ],
          "rows": [
            {
              "instance": "django__django-10554",
              "band": "No panel model solved it",
              "panel": 0,
              "opus": "resolved",
              "sonnet": "unresolved (escalated)",
              "opusCost": 9.99,
              "sonnetCost": 1.52,
              "opusMin": 25.29,
              "sonnetMin": 4.74
            },
            {
              "instance": "matplotlib__matplotlib-21568",
              "band": "No panel model solved it",
              "panel": 0,
              "opus": "resolved (escalated)",
              "sonnet": "resolved",
              "opusCost": 4.9,
              "sonnetCost": 3.02,
              "opusMin": 9.76,
              "sonnetMin": 9.37
            },
            {
              "instance": "django__django-15022",
              "band": "Under half solved it",
              "panel": 4,
              "opus": "unresolved (escalated)",
              "sonnet": "unresolved",
              "opusCost": 7.87,
              "sonnetCost": 4.1,
              "opusMin": 20.26,
              "sonnetMin": 15
            },
            {
              "instance": "django__django-11885",
              "band": "Under half solved it",
              "panel": 3,
              "opus": "not started",
              "sonnet": "resolved",
              "opusCost": null,
              "sonnetCost": 2.27,
              "opusMin": null,
              "sonnetMin": 7.29
            },
            {
              "instance": "django__django-14631",
              "band": "Half or more solved it",
              "panel": 9,
              "opus": "not started",
              "sonnet": "empty patch (escalated)",
              "opusCost": null,
              "sonnetCost": 2.8,
              "opusMin": null,
              "sonnetMin": 38.17
            },
            {
              "instance": "django__django-12193",
              "band": "Half or more solved it",
              "panel": 10,
              "opus": "not started",
              "sonnet": "resolved",
              "opusCost": null,
              "sonnetCost": 2.06,
              "opusMin": null,
              "sonnetMin": 5.74
            },
            {
              "instance": "sympy__sympy-18189",
              "band": "Every panel model solved it",
              "panel": 11,
              "opus": "not started",
              "sonnet": "empty patch (escalated)",
              "opusCost": null,
              "sonnetCost": 0.89,
              "opusMin": null,
              "sonnetMin": 1.63
            },
            {
              "instance": "django__django-11333",
              "band": "Every panel model solved it",
              "panel": 11,
              "opus": "not started",
              "sonnet": "resolved",
              "opusCost": null,
              "sonnetCost": 2.51,
              "opusMin": null,
              "sonnetMin": 7.99
            }
          ]
        }
      ],
      "related": [
        "swe-bench-verified",
        "coding-agents-head-to-head",
        "effort-ladder"
      ],
      "hero": {
        "statIds": [
          "swebench-opus-resolved",
          "swebench-sonnet-resolved"
        ],
        "testStatId": "swebench-opus-sonnet-mcnemar"
      }
    },
    {
      "slug": "blind-review-head-to-head",
      "title": "AI pull requests vs merged human pull requests, judged blind",
      "seoTitle": "AI vs human pull requests: a blind multi-model review",
      "description": "A blind panel of Claude and GPT critics preferred Agent's change over the merged human change on 9 of 12 real tasks. Votes, scores, caveats.",
      "question": "When critics cannot see which change came from a person, do they prefer the AI worker’s pull request or the one the maintainers merged?",
      "answer": "On the latest attempt per task, the blind panel preferred the AI change on 9 of 12 tasks (75%, 95% interval 47% to 91%). On the first scored attempt it was 6 of 12 (25% to 75%). Later attempts learned from earlier ones, so the first-attempt figure is the cleaner estimate. Public OSS tasks: 4 of 5; private Go tasks: 5 of 7. Across 132 single verdicts, 69% preferred the AI change. A critic panel's preference is not a merge decision and not a correctness proof.",
      "date": "2026-09-28",
      "updated": "2026-10-05",
      "tags": [
        "code-review",
        "ai-vs-human",
        "llm-as-judge",
        "pull-requests"
      ],
      "method": [
        "Each task is a real merged pull request: the issue as the ticket, the repository at the base commit, the merged change as the human reference.",
        "The Agent worker solves the ticket in a sandbox without seeing the reference. Its pull request and the merged one form a pair.",
        "Critic models (Claude Opus 5, Claude Fable 5, Claude Sonnet, and on some pairs Codex or GPT 5.5) review both changes without labels, once in each order.",
        "Decision rule 2: the AI change wins on a strict majority of verdicts and no panel veto (serious issues flagged by at least two critics).",
        "Every scored attempt is kept. A re-scoring of the same pair under the current rule replaces the earlier scoring."
      ],
      "caveats": [
        "n = 12 tasks: intervals are wide.",
        "Most critics are Anthropic models, and the worker runs on an Anthropic model. Same-family preference is possible; the per-critic chart shows the OpenAI critics’ share.",
        "Latest attempts are not independent first tries: they had lessons, review replays and some operator answers from earlier attempts on the same task.",
        "7 of the 12 tasks come from one private Go service. They carry neutral labels (private-go-a and so on), and only their language and kind are published.",
        "4 pairs were flagged for position bias (a critic flipped when the order swapped).",
        "The human change was merged, reviewed and shipped. A critic preference says nothing about long-term maintenance cost."
      ],
      "sourceIds": [
        "agent-blind-review"
      ],
      "stats": [
        {
          "id": "ai-preferred-latest",
          "label": "Tasks where the panel preferred the AI change (latest attempt)",
          "value": 0.75,
          "unit": "rate",
          "display": "75% (9/12)",
          "n": 12,
          "ci": [
            0.4677,
            0.9111
          ],
          "note": "Latest attempts come after earlier attempts on the same task; see caveats."
        },
        {
          "id": "ai-preferred-first",
          "label": "Tasks where the panel preferred the AI change (first scored attempt)",
          "value": 0.5,
          "unit": "rate",
          "display": "50% (6/12)",
          "n": 12,
          "ci": [
            0.2538,
            0.7462
          ]
        },
        {
          "id": "ai-preferred-all-pairs",
          "label": "All scored pairs where the panel preferred the AI change",
          "value": 0.6,
          "unit": "rate",
          "display": "60% (12/20)",
          "n": 20,
          "ci": [
            0.3866,
            0.7812
          ]
        },
        {
          "id": "ai-preferred-public",
          "label": "Public OSS tasks, latest attempt",
          "value": 0.8,
          "unit": "rate",
          "display": "80% (4/5)",
          "n": 5,
          "ci": [
            0.3755,
            0.9638
          ]
        },
        {
          "id": "verdicts-ai",
          "label": "Single critic verdicts that preferred the AI change",
          "value": 0.6894,
          "unit": "rate",
          "display": "69% (91/132)",
          "n": 132,
          "ci": [
            0.606,
            0.762
          ]
        },
        {
          "id": "review-spend",
          "label": "Notional spend: worker runs + critic panel",
          "value": 1310.5,
          "unit": "usd",
          "display": "$979 + $332",
          "n": 20,
          "note": "List-price estimates of subscription calls over all scored pairs."
        },
        {
          "id": "position-bias-flags",
          "label": "Pairs flagged for position bias",
          "value": 4,
          "unit": "count",
          "display": "4 of 20",
          "n": 20,
          "note": "A critic model flipped its preference when the two changes swapped places."
        }
      ],
      "charts": [
        {
          "id": "blind-review-first-vs-latest",
          "title": "First attempt vs latest attempt",
          "subtitle": "Share of tasks where the blind panel preferred the AI change",
          "kind": "dot-range",
          "unit": "rate",
          "yLabel": "Tasks preferred",
          "series": [
            {
              "name": "AI preferred",
              "points": [
                {
                  "label": "First scored attempt",
                  "value": 0.5,
                  "lo": 0.2538,
                  "hi": 0.7462,
                  "n": 12
                },
                {
                  "label": "Latest attempt",
                  "value": 0.75,
                  "lo": 0.4677,
                  "hi": 0.9111,
                  "n": 12,
                  "highlight": true
                },
                {
                  "label": "Public OSS tasks, latest",
                  "value": 0.8,
                  "lo": 0.3755,
                  "hi": 0.9638,
                  "n": 5
                },
                {
                  "label": "Private tasks, latest",
                  "value": 0.7143,
                  "lo": 0.3589,
                  "hi": 0.9178,
                  "n": 7
                }
              ]
            }
          ],
          "note": "Later attempts had the earlier attempts’ lessons, review replays and, on some tasks, operator answers. They are not independent first tries.",
          "sourceIds": [
            "agent-blind-review"
          ]
        },
        {
          "id": "blind-review-votes-by-task",
          "title": "Blind panel votes per task (latest attempt)",
          "subtitle": "Each critic judges both orders without labels",
          "kind": "stacked-bar",
          "unit": "count",
          "yLabel": "Verdicts",
          "series": [
            {
              "name": "Prefers AI change",
              "points": [
                {
                  "label": "h3js/h3#1533",
                  "value": 8,
                  "n": 8,
                  "highlight": true
                },
                {
                  "label": "private-go-a (private Go)",
                  "value": 8,
                  "n": 8,
                  "highlight": true
                },
                {
                  "label": "open-telemetry/opentelemetry-go#8706",
                  "value": 8,
                  "n": 8,
                  "highlight": true
                },
                {
                  "label": "fastify/session#348",
                  "value": 8,
                  "n": 8,
                  "highlight": true
                },
                {
                  "label": "private-go-d (private Go)",
                  "value": 6,
                  "n": 6,
                  "highlight": true
                },
                {
                  "label": "redis/go-redis#3914",
                  "value": 8,
                  "n": 8,
                  "highlight": true
                },
                {
                  "label": "private-go-g (private Go)",
                  "value": 6,
                  "n": 6,
                  "highlight": true
                },
                {
                  "label": "private-go-f (private Go)",
                  "value": 6,
                  "n": 6,
                  "highlight": true
                },
                {
                  "label": "private-go-e (private Go)",
                  "value": 6,
                  "n": 6,
                  "highlight": true
                },
                {
                  "label": "aio-libs/aiohttp#13122",
                  "value": 4,
                  "n": 8,
                  "highlight": false
                },
                {
                  "label": "private-go-c (private Go)",
                  "value": 0,
                  "n": 6,
                  "highlight": false
                },
                {
                  "label": "private-go-b (private Go)",
                  "value": 0,
                  "n": 6,
                  "highlight": false
                }
              ]
            },
            {
              "name": "Prefers human change",
              "points": [
                {
                  "label": "h3js/h3#1533",
                  "value": 0,
                  "n": 8
                },
                {
                  "label": "private-go-a (private Go)",
                  "value": 0,
                  "n": 8
                },
                {
                  "label": "open-telemetry/opentelemetry-go#8706",
                  "value": 0,
                  "n": 8
                },
                {
                  "label": "fastify/session#348",
                  "value": 0,
                  "n": 8
                },
                {
                  "label": "private-go-d (private Go)",
                  "value": 0,
                  "n": 6
                },
                {
                  "label": "redis/go-redis#3914",
                  "value": 0,
                  "n": 8
                },
                {
                  "label": "private-go-g (private Go)",
                  "value": 0,
                  "n": 6
                },
                {
                  "label": "private-go-f (private Go)",
                  "value": 0,
                  "n": 6
                },
                {
                  "label": "private-go-e (private Go)",
                  "value": 0,
                  "n": 6
                },
                {
                  "label": "aio-libs/aiohttp#13122",
                  "value": 4,
                  "n": 8
                },
                {
                  "label": "private-go-c (private Go)",
                  "value": 6,
                  "n": 6
                },
                {
                  "label": "private-go-b (private Go)",
                  "value": 6,
                  "n": 6
                }
              ]
            },
            {
              "name": "Tie",
              "points": [
                {
                  "label": "h3js/h3#1533",
                  "value": 0,
                  "n": 8
                },
                {
                  "label": "private-go-a (private Go)",
                  "value": 0,
                  "n": 8
                },
                {
                  "label": "open-telemetry/opentelemetry-go#8706",
                  "value": 0,
                  "n": 8
                },
                {
                  "label": "fastify/session#348",
                  "value": 0,
                  "n": 8
                },
                {
                  "label": "private-go-d (private Go)",
                  "value": 0,
                  "n": 6
                },
                {
                  "label": "redis/go-redis#3914",
                  "value": 0,
                  "n": 8
                },
                {
                  "label": "private-go-g (private Go)",
                  "value": 0,
                  "n": 6
                },
                {
                  "label": "private-go-f (private Go)",
                  "value": 0,
                  "n": 6
                },
                {
                  "label": "private-go-e (private Go)",
                  "value": 0,
                  "n": 6
                },
                {
                  "label": "aio-libs/aiohttp#13122",
                  "value": 0,
                  "n": 8
                },
                {
                  "label": "private-go-c (private Go)",
                  "value": 0,
                  "n": 6
                },
                {
                  "label": "private-go-b (private Go)",
                  "value": 0,
                  "n": 6
                }
              ]
            }
          ],
          "note": "The human change is the one the maintainers merged upstream. Private tasks are from one private Go service and carry neutral labels.",
          "sourceIds": [
            "agent-blind-review"
          ]
        },
        {
          "id": "blind-review-dimension-scores",
          "title": "What the critics scored higher",
          "subtitle": "Mean 1-5 score per dimension, latest attempt of 12 tasks",
          "kind": "grouped-bar",
          "unit": "score",
          "yLabel": "Mean score (1-5)",
          "series": [
            {
              "name": "AI change",
              "points": [
                {
                  "label": "Correctness",
                  "value": 4.4,
                  "n": 12,
                  "highlight": true
                },
                {
                  "label": "Root cause",
                  "value": 4.46,
                  "n": 12,
                  "highlight": true
                },
                {
                  "label": "Edge cases",
                  "value": 3.95,
                  "n": 12,
                  "highlight": true
                },
                {
                  "label": "Tests",
                  "value": 4.51,
                  "n": 12,
                  "highlight": true
                },
                {
                  "label": "Completeness",
                  "value": 4.22,
                  "n": 12,
                  "highlight": true
                },
                {
                  "label": "Compatibility",
                  "value": 4.39,
                  "n": 12,
                  "highlight": true
                },
                {
                  "label": "Security",
                  "value": 4.38,
                  "n": 12,
                  "highlight": true
                },
                {
                  "label": "Maintainability",
                  "value": 4.19,
                  "n": 12,
                  "highlight": true
                },
                {
                  "label": "Simplicity",
                  "value": 4.19,
                  "n": 12,
                  "highlight": true
                },
                {
                  "label": "Conventions",
                  "value": 4.36,
                  "n": 12,
                  "highlight": true
                },
                {
                  "label": "Docs",
                  "value": 4.04,
                  "n": 12,
                  "highlight": true
                },
                {
                  "label": "Communication",
                  "value": 4.59,
                  "n": 12,
                  "highlight": true
                },
                {
                  "label": "Production readiness",
                  "value": 4.03,
                  "n": 12,
                  "highlight": true
                },
                {
                  "label": "Overall",
                  "value": 4.14,
                  "n": 12,
                  "highlight": true
                }
              ]
            },
            {
              "name": "Merged human change",
              "points": [
                {
                  "label": "Correctness",
                  "value": 3.86,
                  "n": 12
                },
                {
                  "label": "Root cause",
                  "value": 3.97,
                  "n": 12
                },
                {
                  "label": "Edge cases",
                  "value": 3.49,
                  "n": 12
                },
                {
                  "label": "Tests",
                  "value": 3.29,
                  "n": 12
                },
                {
                  "label": "Completeness",
                  "value": 3.34,
                  "n": 12
                },
                {
                  "label": "Compatibility",
                  "value": 3.74,
                  "n": 12
                },
                {
                  "label": "Security",
                  "value": 4.14,
                  "n": 12
                },
                {
                  "label": "Maintainability",
                  "value": 3.64,
                  "n": 12
                },
                {
                  "label": "Simplicity",
                  "value": 3.85,
                  "n": 12
                },
                {
                  "label": "Conventions",
                  "value": 3.52,
                  "n": 12
                },
                {
                  "label": "Docs",
                  "value": 3.2,
                  "n": 12
                },
                {
                  "label": "Communication",
                  "value": 3.58,
                  "n": 12
                },
                {
                  "label": "Production readiness",
                  "value": 3.21,
                  "n": 12
                },
                {
                  "label": "Overall",
                  "value": 3.2,
                  "n": 12
                }
              ]
            }
          ],
          "note": "Unweighted mean over tasks of each pair’s mean critic score. Critics were not told which change came from a person.",
          "sourceIds": [
            "agent-blind-review"
          ]
        },
        {
          "id": "blind-review-critic-agreement",
          "title": "Does the judge’s model family matter?",
          "subtitle": "Share of single verdicts preferring the AI change, per critic model, all scored pairs",
          "kind": "dot-range",
          "unit": "rate",
          "yLabel": "Prefers AI change",
          "series": [
            {
              "name": "Critic model",
              "points": [
                {
                  "label": "GPT 5.5 (OpenAI)",
                  "value": 1,
                  "lo": 0.3424,
                  "hi": 1,
                  "n": 2
                },
                {
                  "label": "Codex (GPT) (OpenAI)",
                  "value": 0.8,
                  "lo": 0.4902,
                  "hi": 0.9433,
                  "n": 10
                },
                {
                  "label": "Claude Opus 5 (Anthropic)",
                  "value": 0.7,
                  "lo": 0.5457,
                  "hi": 0.8193,
                  "n": 40
                },
                {
                  "label": "Claude Sonnet (Anthropic)",
                  "value": 0.675,
                  "lo": 0.5202,
                  "hi": 0.7992,
                  "n": 40
                },
                {
                  "label": "Claude Fable 5 (Anthropic)",
                  "value": 0.65,
                  "lo": 0.4951,
                  "hi": 0.7787,
                  "n": 40
                }
              ]
            }
          ],
          "note": "Whiskers are 95% Wilson intervals over single verdicts (verdicts within one pair are not independent). The worker model is Anthropic; OpenAI critics joined later and judged fewer pairs.",
          "sourceIds": [
            "agent-blind-review"
          ]
        }
      ],
      "tables": [
        {
          "id": "blind-review-pairs",
          "title": "Every scored pair",
          "columns": [
            {
              "key": "task",
              "label": "Task",
              "unit": "text"
            },
            {
              "key": "language",
              "label": "Language",
              "unit": "text"
            },
            {
              "key": "category",
              "label": "Kind",
              "unit": "text"
            },
            {
              "key": "iteration",
              "label": "Attempt",
              "unit": "count"
            },
            {
              "key": "outcome",
              "label": "Panel decision",
              "unit": "text"
            },
            {
              "key": "votes",
              "label": "Votes AI-human",
              "unit": "text"
            },
            {
              "key": "critics",
              "label": "Critics",
              "unit": "text"
            },
            {
              "key": "answers",
              "label": "Operator answers",
              "unit": "count"
            },
            {
              "key": "runUsd",
              "label": "Run cost (notional)",
              "unit": "usd"
            }
          ],
          "rows": [
            {
              "task": "aio-libs/aiohttp#13122",
              "language": "Python",
              "category": "feature",
              "iteration": 3,
              "outcome": "Human preferred",
              "votes": "4-4",
              "critics": "Claude Opus 5, Codex (GPT), Claude Fable 5, Claude Sonnet",
              "answers": 5,
              "runUsd": 78.28
            },
            {
              "task": "redis/go-redis#3914",
              "language": "Go",
              "category": "ambiguous",
              "iteration": 1,
              "outcome": "Human preferred",
              "votes": "1-5",
              "critics": "Claude Opus 5, Claude Fable 5, Claude Sonnet",
              "answers": 0,
              "runUsd": 21.05
            },
            {
              "task": "redis/go-redis#3914",
              "language": "Go",
              "category": "ambiguous",
              "iteration": 2,
              "outcome": "AI preferred",
              "votes": "8-0",
              "critics": "Claude Opus 5, Codex (GPT), Claude Fable 5, Claude Sonnet",
              "answers": 1,
              "runUsd": 31.5
            },
            {
              "task": "h3js/h3#1533",
              "language": "TypeScript",
              "category": "refactor",
              "iteration": 2,
              "outcome": "AI preferred",
              "votes": "8-0",
              "critics": "Claude Opus 5, Codex (GPT), Claude Fable 5, Claude Sonnet",
              "answers": 0,
              "runUsd": 21.77
            },
            {
              "task": "open-telemetry/opentelemetry-go#8706",
              "language": "Go",
              "category": "bug",
              "iteration": 1,
              "outcome": "Human preferred",
              "votes": "3-3",
              "critics": "Claude Opus 5, Claude Fable 5, Claude Sonnet",
              "answers": 2,
              "runUsd": 35.17
            },
            {
              "task": "open-telemetry/opentelemetry-go#8706",
              "language": "Go",
              "category": "bug",
              "iteration": 2,
              "outcome": "AI preferred",
              "votes": "8-0",
              "critics": "Claude Opus 5, Codex (GPT), Claude Fable 5, Claude Sonnet",
              "answers": 3,
              "runUsd": 80.48
            },
            {
              "task": "private-go-a (private Go)",
              "language": "Go",
              "category": "bug",
              "iteration": 1,
              "outcome": "AI preferred",
              "votes": "6-0",
              "critics": "Claude Opus 5, Claude Fable 5, Claude Sonnet",
              "answers": 1,
              "runUsd": 72.29
            },
            {
              "task": "private-go-a (private Go)",
              "language": "Go",
              "category": "bug",
              "iteration": 3,
              "outcome": "AI preferred",
              "votes": "6-0",
              "critics": "Claude Opus 5, Claude Fable 5, Claude Sonnet",
              "answers": 3,
              "runUsd": 54.51
            },
            {
              "task": "private-go-a (private Go)",
              "language": "Go",
              "category": "bug",
              "iteration": 7,
              "outcome": "AI preferred",
              "votes": "8-0",
              "critics": "Claude Opus 5, Claude Fable 5, GPT 5.5, Claude Sonnet",
              "answers": 0,
              "runUsd": 47.65
            },
            {
              "task": "private-go-b (private Go)",
              "language": "Go",
              "category": "bug",
              "iteration": 1,
              "outcome": "Human preferred",
              "votes": "1-5",
              "critics": "Claude Opus 5, Claude Fable 5, Claude Sonnet",
              "answers": 0,
              "runUsd": 74.97
            },
            {
              "task": "private-go-b (private Go)",
              "language": "Go",
              "category": "bug",
              "iteration": 2,
              "outcome": "Human preferred",
              "votes": "0-6",
              "critics": "Claude Opus 5, Claude Fable 5, Claude Sonnet",
              "answers": 2,
              "runUsd": 74.76
            },
            {
              "task": "private-go-b (private Go)",
              "language": "Go",
              "category": "bug",
              "iteration": 3,
              "outcome": "Human preferred",
              "votes": "0-6",
              "critics": "Claude Opus 5, Claude Fable 5, Claude Sonnet",
              "answers": 4,
              "runUsd": 70.28
            },
            {
              "task": "private-go-c (private Go)",
              "language": "Go",
              "category": "bug",
              "iteration": 1,
              "outcome": "Human preferred",
              "votes": "0-6",
              "critics": "Claude Opus 5, Claude Fable 5, Claude Sonnet",
              "answers": 3,
              "runUsd": 73.13
            },
            {
              "task": "private-go-d (private Go)",
              "language": "Go",
              "category": "refactor",
              "iteration": 1,
              "outcome": "AI preferred",
              "votes": "6-0",
              "critics": "Claude Opus 5, Claude Fable 5, Claude Sonnet",
              "answers": 3,
              "runUsd": 43.39
            },
            {
              "task": "private-go-e (private Go)",
              "language": "Go",
              "category": "feature",
              "iteration": 1,
              "outcome": "Human preferred",
              "votes": "0-6",
              "critics": "Claude Opus 5, Claude Fable 5, Claude Sonnet",
              "answers": 2,
              "runUsd": 22.93
            },
            {
              "task": "private-go-e (private Go)",
              "language": "Go",
              "category": "feature",
              "iteration": 2,
              "outcome": "AI preferred",
              "votes": "6-0",
              "critics": "Claude Opus 5, Claude Fable 5, Claude Sonnet",
              "answers": 1,
              "runUsd": 32.75
            },
            {
              "task": "private-go-e (private Go)",
              "language": "Go",
              "category": "feature",
              "iteration": 3,
              "outcome": "AI preferred",
              "votes": "6-0",
              "critics": "Claude Opus 5, Claude Fable 5, Claude Sonnet",
              "answers": 3,
              "runUsd": 26.81
            },
            {
              "task": "private-go-f (private Go)",
              "language": "Go",
              "category": "feature",
              "iteration": 1,
              "outcome": "AI preferred",
              "votes": "6-0",
              "critics": "Claude Opus 5, Claude Fable 5, Claude Sonnet",
              "answers": 2,
              "runUsd": 53.93
            },
            {
              "task": "private-go-g (private Go)",
              "language": "Go",
              "category": "feature",
              "iteration": 2,
              "outcome": "AI preferred",
              "votes": "6-0",
              "critics": "Claude Opus 5, Claude Fable 5, Claude Sonnet",
              "answers": 1,
              "runUsd": 41.35
            },
            {
              "task": "fastify/session#348",
              "language": "JavaScript",
              "category": "security",
              "iteration": 1,
              "outcome": "AI preferred",
              "votes": "8-0",
              "critics": "Claude Opus 5, Codex (GPT), Claude Fable 5, Claude Sonnet",
              "answers": 0,
              "runUsd": 21.53
            }
          ]
        }
      ]
    },
    {
      "slug": "model-head-to-head",
      "title": "Haiku vs Sonnet vs Opus vs Fable vs Codex: a timed head-to-head",
      "seoTitle": "Haiku vs Sonnet vs Opus vs Fable vs Codex: speed and tokens",
      "description": "130 timed calls on five validated tasks: Claude Haiku, Sonnet, Opus and Fable against Codex GPT-6.1 Sol. Pass rate, speed, tokens, cost per pass.",
      "question": "On short tasks with strict validators, how do the Claude Code models and efforts compare with Codex on pass rate, speed and tokens?",
      "answer": "127 of 130 calls passed (98%, 95% interval 93% to 99%), so on these short tasks pass rate barely separates the configurations (non-passes: Claude Sonnet 5.5 · Claude Code on Multi-step shift arithmetic ×3 (\"Output does not exactly match the expected text\")). Every non-pass ended on the expected answer (2292) but added working lines, which the exact-text validator rejects by design: a format miss, not a wrong answer. Speed separates them more: the fastest configuration was Claude Fable 5.1 · Claude Code at a median 1.9 s per call, the slowest GPT-6.1 Sol (low) · Codex CLI at 6.3 s. Claude Code medians ran from 1.9 s to 4.4 s. Codex CLI medians ran from 5.6 s to 6.3 s. Single calls vary a lot (see the ranges), so neighbouring configurations are not separated. Codex CLI sends a median 12,124 input tokens per call against 2,130 for Claude Code, mostly the CLI’s own context. At list price (a calculation; the calls ran on subscriptions), the cheapest passing answer came from Claude Sonnet 5.5 · Claude Code at $0.0062. On speed, cost per pass and pass rate together, the frontier is Claude Fable 5.1 · Claude Code, Claude Sonnet 5.5 · Claude Code, Claude Opus 5.5 (high) · Claude Code, Claude Opus 5.5 · Claude Code, Claude Opus 5.5 (low) · Claude Code; with 10 to 15 calls per configuration, small gaps inside it are within the spread of single calls. A harder follow-up with eight tasks and strict validators: /benchmarks/hard-model-head-to-head.",
      "date": "2026-10-05",
      "updated": "2026-10-05",
      "tags": [
        "head-to-head",
        "claude-haiku",
        "claude-sonnet",
        "claude-opus",
        "claude-fable",
        "codex",
        "latency"
      ],
      "method": [
        "Protocol declared before the first call: 5 cases with deterministic validators (behavioral checks in a sandbox, canonical JSON or exact text).",
        "Claude Code: Haiku, Sonnet, Opus and Fable at default effort, plus Opus at low and high, 3 repetitions. Codex: GPT-6.1 Sol at low, medium and high, 2 repetitions.",
        "Isolation: fresh empty working folder, tools off, no MCP servers, no session persistence, one turn. One call at a time per account.",
        "Every attempt is kept. Nothing is retried. A cell that did not run is shown as trimmed.",
        "All 130 planned calls ran. Nothing was trimmed or retried, and no run hit a usage or rate limit.",
        "Cost per passing answer: list price × reported tokens for every call in the configuration (cache reads and writes priced separately), divided by its passes."
      ],
      "caveats": [
        "The tasks are short and easy; pass rate saturates. Latency and tokens carry the signal. A harder follow-up with eight tasks and strict validators: /benchmarks/hard-model-head-to-head.",
        "Few repetitions per cell (2 or 3 per task). Medians with ranges, not intervals.",
        "CLI timings include CLI start-up and the CLI’s own system prompt.",
        "One host, one network, one day.",
        "List-price costs are calculations; the calls used flat subscriptions.",
        "The prompt cache stayed at the provider default, so cache counters differ by route and by call order.",
        "Haiku 4.5 reported reasoning tokens on most calls under the CLI default, which explains much of its extra time and output.",
        "Haiku and Fable are not in the platform runner catalog, so these calls used the runner’s CLI functions directly; each receipt records the model the CLI reported."
      ],
      "sourceIds": [
        "agent-provider-h2h",
        "calc-repricing",
        "price-anthropic",
        "price-openai"
      ],
      "stats": [
        {
          "id": "h2h-pass-all",
          "label": "Calls that passed their validator",
          "value": 0.9769,
          "unit": "rate",
          "display": "98% (127/130)",
          "n": 130,
          "ci": [
            0.9343,
            0.9921
          ]
        },
        {
          "id": "h2h-configs",
          "label": "Configurations compared",
          "value": 9,
          "unit": "count",
          "display": "9"
        },
        {
          "id": "h2h-fastest",
          "label": "Fastest configuration (median total time)",
          "value": 1.94,
          "unit": "seconds",
          "display": "Claude Fable 5.1 · Claude Code: 1.9 s",
          "n": 15
        },
        {
          "id": "h2h-slowest",
          "label": "Slowest configuration (median total time)",
          "value": 6.26,
          "unit": "seconds",
          "display": "GPT-6.1 Sol (low) · Codex CLI: 6.3 s",
          "n": 10
        },
        {
          "id": "h2h-cheapest-per-pass",
          "label": "Lowest list-price cost per passing answer (calculation)",
          "value": 0.00624,
          "unit": "usd",
          "display": "Claude Sonnet 5.5 · Claude Code: $0.0062",
          "n": 15
        },
        {
          "id": "h2h-codex-input-tokens",
          "label": "Median input tokens per call, Codex CLI vs Claude Code",
          "value": 12124,
          "unit": "tokens",
          "display": "12,124 vs 2,130",
          "n": 130
        }
      ],
      "charts": [
        {
          "id": "h2h-pass-rate",
          "title": "Pass rate on five validated tasks",
          "subtitle": "Every call counts; failures and timeouts are non-passes",
          "kind": "dot-range",
          "unit": "rate",
          "yLabel": "Passed",
          "series": [
            {
              "name": "Pass rate",
              "points": [
                {
                  "label": "Claude Fable 5.1 · Claude Code",
                  "value": 1,
                  "lo": 0.7961,
                  "hi": 1,
                  "n": 15
                },
                {
                  "label": "Claude Sonnet 5.5 · Claude Code",
                  "value": 0.8,
                  "lo": 0.5481,
                  "hi": 0.9295,
                  "n": 15
                },
                {
                  "label": "Claude Opus 5.5 (high) · Claude Code",
                  "value": 1,
                  "lo": 0.7961,
                  "hi": 1,
                  "n": 15
                },
                {
                  "label": "Claude Opus 5.5 · Claude Code",
                  "value": 1,
                  "lo": 0.7961,
                  "hi": 1,
                  "n": 15
                },
                {
                  "label": "Claude Opus 5.5 (low) · Claude Code",
                  "value": 1,
                  "lo": 0.7961,
                  "hi": 1,
                  "n": 15
                },
                {
                  "label": "Claude Haiku 4.5 · Claude Code",
                  "value": 1,
                  "lo": 0.7961,
                  "hi": 1,
                  "n": 15
                },
                {
                  "label": "GPT-6.1 Sol (high) · Codex CLI",
                  "value": 1,
                  "lo": 0.7961,
                  "hi": 1,
                  "n": 15
                },
                {
                  "label": "GPT-6.1 Sol (medium) · Codex CLI",
                  "value": 1,
                  "lo": 0.7961,
                  "hi": 1,
                  "n": 15
                },
                {
                  "label": "GPT-6.1 Sol (low) · Codex CLI",
                  "value": 1,
                  "lo": 0.7225,
                  "hi": 1,
                  "n": 10
                }
              ]
            }
          ],
          "note": "Whiskers are 95% Wilson intervals. The tasks are short and most configurations pass nearly all of them, so pass rate does not separate the models here.",
          "sourceIds": [
            "agent-provider-h2h"
          ]
        },
        {
          "id": "h2h-total-latency",
          "title": "Total time per call",
          "subtitle": "Median per configuration; whiskers = fastest and slowest call",
          "kind": "dot-range",
          "unit": "seconds",
          "yLabel": "Seconds",
          "series": [
            {
              "name": "Total time per call",
              "points": [
                {
                  "label": "Claude Fable 5.1 · Claude Code",
                  "value": 1.94,
                  "lo": 1.41,
                  "hi": 9.83,
                  "n": 15,
                  "highlight": true
                },
                {
                  "label": "Claude Sonnet 5.5 · Claude Code",
                  "value": 2.31,
                  "lo": 2.17,
                  "hi": 7.73,
                  "n": 15,
                  "highlight": true
                },
                {
                  "label": "Claude Opus 5.5 (high) · Claude Code",
                  "value": 2.71,
                  "lo": 2.45,
                  "hi": 11.78,
                  "n": 15,
                  "highlight": true
                },
                {
                  "label": "Claude Opus 5.5 · Claude Code",
                  "value": 2.75,
                  "lo": 2.47,
                  "hi": 8.91,
                  "n": 15,
                  "highlight": true
                },
                {
                  "label": "Claude Opus 5.5 (low) · Claude Code",
                  "value": 2.83,
                  "lo": 2.35,
                  "hi": 6.62,
                  "n": 15,
                  "highlight": true
                },
                {
                  "label": "Claude Haiku 4.5 · Claude Code",
                  "value": 4.43,
                  "lo": 3.16,
                  "hi": 23.57,
                  "n": 15,
                  "highlight": true
                },
                {
                  "label": "GPT-6.1 Sol (high) · Codex CLI",
                  "value": 5.6,
                  "lo": 4.05,
                  "hi": 19.52,
                  "n": 15,
                  "highlight": false
                },
                {
                  "label": "GPT-6.1 Sol (medium) · Codex CLI",
                  "value": 5.65,
                  "lo": 4.1,
                  "hi": 25.46,
                  "n": 15,
                  "highlight": false
                },
                {
                  "label": "GPT-6.1 Sol (low) · Codex CLI",
                  "value": 6.26,
                  "lo": 4.65,
                  "hi": 10.47,
                  "n": 10,
                  "highlight": false
                }
              ]
            }
          ],
          "note": "One host, one network, one day. Whiskers are a range, not a confidence interval.",
          "sourceIds": [
            "agent-provider-h2h"
          ]
        },
        {
          "id": "h2h-first-useful-latency",
          "title": "Time to first useful output",
          "subtitle": "Median per configuration; whiskers = fastest and slowest call",
          "kind": "dot-range",
          "unit": "seconds",
          "yLabel": "Seconds",
          "series": [
            {
              "name": "Time to first useful output",
              "points": [
                {
                  "label": "Claude Fable 5.1 · Claude Code",
                  "value": 1.2,
                  "lo": 0.95,
                  "hi": 7.9,
                  "n": 15,
                  "highlight": true
                },
                {
                  "label": "Claude Sonnet 5.5 · Claude Code",
                  "value": 1.56,
                  "lo": 0.99,
                  "hi": 6.39,
                  "n": 15,
                  "highlight": true
                },
                {
                  "label": "Claude Opus 5.5 (high) · Claude Code",
                  "value": 2.04,
                  "lo": 1.4,
                  "hi": 9.94,
                  "n": 15,
                  "highlight": true
                },
                {
                  "label": "Claude Opus 5.5 · Claude Code",
                  "value": 1.92,
                  "lo": 1.56,
                  "hi": 7.23,
                  "n": 15,
                  "highlight": true
                },
                {
                  "label": "Claude Opus 5.5 (low) · Claude Code",
                  "value": 2.39,
                  "lo": 1.45,
                  "hi": 4.9,
                  "n": 15,
                  "highlight": true
                },
                {
                  "label": "Claude Haiku 4.5 · Claude Code",
                  "value": 3.63,
                  "lo": 2.78,
                  "hi": 22.27,
                  "n": 15,
                  "highlight": true
                },
                {
                  "label": "GPT-6.1 Sol (high) · Codex CLI",
                  "value": 5.32,
                  "lo": 3.64,
                  "hi": 16.37,
                  "n": 15,
                  "highlight": false
                },
                {
                  "label": "GPT-6.1 Sol (medium) · Codex CLI",
                  "value": 5.05,
                  "lo": 3.36,
                  "hi": 17.82,
                  "n": 15,
                  "highlight": false
                },
                {
                  "label": "GPT-6.1 Sol (low) · Codex CLI",
                  "value": 5.14,
                  "lo": 4.02,
                  "hi": 8.5,
                  "n": 10,
                  "highlight": false
                }
              ]
            }
          ],
          "note": "One host, one network, one day. Whiskers are a range, not a confidence interval.",
          "sourceIds": [
            "agent-provider-h2h"
          ]
        },
        {
          "id": "h2h-input-tokens",
          "title": "Input tokens per call: what the CLI sends",
          "subtitle": "Mean per call, split into prompt-cache reads and other input",
          "kind": "stacked-bar",
          "unit": "tokens",
          "yLabel": "Tokens",
          "series": [
            {
              "name": "Cache read",
              "points": [
                {
                  "label": "Claude Fable 5.1 · Claude Code",
                  "value": 2760,
                  "n": 15
                },
                {
                  "label": "Claude Sonnet 5.5 · Claude Code",
                  "value": 1401,
                  "n": 15
                },
                {
                  "label": "Claude Opus 5.5 (high) · Claude Code",
                  "value": 1463,
                  "n": 15
                },
                {
                  "label": "Claude Opus 5.5 · Claude Code",
                  "value": 1401,
                  "n": 15
                },
                {
                  "label": "Claude Opus 5.5 (low) · Claude Code",
                  "value": 1463,
                  "n": 15
                },
                {
                  "label": "Claude Haiku 4.5 · Claude Code",
                  "value": 0,
                  "n": 15
                },
                {
                  "label": "GPT-6.1 Sol (high) · Codex CLI",
                  "value": 6716,
                  "n": 15
                },
                {
                  "label": "GPT-6.1 Sol (medium) · Codex CLI",
                  "value": 5180,
                  "n": 15
                },
                {
                  "label": "GPT-6.1 Sol (low) · Codex CLI",
                  "value": 8064,
                  "n": 10
                }
              ]
            },
            {
              "name": "Other input",
              "points": [
                {
                  "label": "Claude Fable 5.1 · Claude Code",
                  "value": 473,
                  "n": 15
                },
                {
                  "label": "Claude Sonnet 5.5 · Claude Code",
                  "value": 685,
                  "n": 15
                },
                {
                  "label": "Claude Opus 5.5 (high) · Claude Code",
                  "value": 619,
                  "n": 15
                },
                {
                  "label": "Claude Opus 5.5 · Claude Code",
                  "value": 680,
                  "n": 15
                },
                {
                  "label": "Claude Opus 5.5 (low) · Claude Code",
                  "value": 618,
                  "n": 15
                },
                {
                  "label": "Claude Haiku 4.5 · Claude Code",
                  "value": 3790,
                  "n": 15
                },
                {
                  "label": "GPT-6.1 Sol (high) · Codex CLI",
                  "value": 5406,
                  "n": 15
                },
                {
                  "label": "GPT-6.1 Sol (medium) · Codex CLI",
                  "value": 6943,
                  "n": 15
                },
                {
                  "label": "GPT-6.1 Sol (low) · Codex CLI",
                  "value": 4059,
                  "n": 10
                }
              ]
            }
          ],
          "note": "The prompts are a few hundred tokens; most input is the CLI’s own system prompt and tool context. Input counts include cache reads, as the vendors report them.",
          "sourceIds": [
            "agent-provider-h2h"
          ]
        },
        {
          "id": "h2h-output-tokens",
          "title": "Output tokens per call",
          "subtitle": "Median per configuration; reasoning tokens where the CLI reports them",
          "kind": "grouped-bar",
          "unit": "tokens",
          "yLabel": "Tokens",
          "series": [
            {
              "name": "Output tokens",
              "points": [
                {
                  "label": "Claude Fable 5.1 · Claude Code",
                  "value": 64,
                  "n": 15
                },
                {
                  "label": "Claude Sonnet 5.5 · Claude Code",
                  "value": 107,
                  "n": 15
                },
                {
                  "label": "Claude Opus 5.5 (high) · Claude Code",
                  "value": 78,
                  "n": 15
                },
                {
                  "label": "Claude Opus 5.5 · Claude Code",
                  "value": 64,
                  "n": 15
                },
                {
                  "label": "Claude Opus 5.5 (low) · Claude Code",
                  "value": 64,
                  "n": 15
                },
                {
                  "label": "Claude Haiku 4.5 · Claude Code",
                  "value": 367,
                  "n": 15
                },
                {
                  "label": "GPT-6.1 Sol (high) · Codex CLI",
                  "value": 42,
                  "n": 15
                },
                {
                  "label": "GPT-6.1 Sol (medium) · Codex CLI",
                  "value": 42,
                  "n": 15
                },
                {
                  "label": "GPT-6.1 Sol (low) · Codex CLI",
                  "value": 42,
                  "n": 10
                }
              ]
            },
            {
              "name": "Reasoning tokens",
              "points": [
                {
                  "label": "Claude Fable 5.1 · Claude Code",
                  "value": 0,
                  "n": 15
                },
                {
                  "label": "Claude Sonnet 5.5 · Claude Code",
                  "value": 0,
                  "n": 15
                },
                {
                  "label": "Claude Opus 5.5 (high) · Claude Code",
                  "value": 34,
                  "n": 15
                },
                {
                  "label": "Claude Opus 5.5 · Claude Code",
                  "value": 0,
                  "n": 15
                },
                {
                  "label": "Claude Opus 5.5 (low) · Claude Code",
                  "value": 0,
                  "n": 15
                },
                {
                  "label": "Claude Haiku 4.5 · Claude Code",
                  "value": 297,
                  "n": 15
                },
                {
                  "label": "GPT-6.1 Sol (high) · Codex CLI",
                  "value": 21,
                  "n": 15
                },
                {
                  "label": "GPT-6.1 Sol (medium) · Codex CLI",
                  "value": 20,
                  "n": 15
                },
                {
                  "label": "GPT-6.1 Sol (low) · Codex CLI",
                  "value": 20,
                  "n": 10
                }
              ]
            }
          ],
          "note": "Vendors count reasoning differently; compare within a vendor first. A median of 0 can mean the CLI did not report reasoning for most calls.",
          "sourceIds": [
            "agent-provider-h2h"
          ]
        },
        {
          "id": "h2h-list-price-per-call",
          "title": "List-price cost per call (calculation)",
          "subtitle": "Reported tokens × list price; the calls ran on subscriptions",
          "kind": "dot-range",
          "unit": "usd",
          "yLabel": "USD per call",
          "series": [
            {
              "name": "Cost per call",
              "points": [
                {
                  "label": "Claude Fable 5.1 · Claude Code",
                  "value": 0.00987,
                  "lo": 0.0049,
                  "hi": 0.05843,
                  "n": 15,
                  "highlight": true
                },
                {
                  "label": "Claude Sonnet 5.5 · Claude Code",
                  "value": 0.0036,
                  "lo": 0.00342,
                  "hi": 0.01021,
                  "n": 15,
                  "highlight": true
                },
                {
                  "label": "Claude Opus 5.5 (high) · Claude Code",
                  "value": 0.00694,
                  "lo": 0.00592,
                  "hi": 0.02708,
                  "n": 15,
                  "highlight": true
                },
                {
                  "label": "Claude Opus 5.5 · Claude Code",
                  "value": 0.00688,
                  "lo": 0.00592,
                  "hi": 0.02226,
                  "n": 15,
                  "highlight": true
                },
                {
                  "label": "Claude Opus 5.5 (low) · Claude Code",
                  "value": 0.00688,
                  "lo": 0.00582,
                  "hi": 0.01793,
                  "n": 15,
                  "highlight": true
                },
                {
                  "label": "Claude Haiku 4.5 · Claude Code",
                  "value": 0.00566,
                  "lo": 0.00513,
                  "hi": 0.01804,
                  "n": 15,
                  "highlight": true
                },
                {
                  "label": "GPT-6.1 Sol (high) · Codex CLI",
                  "value": 0.01047,
                  "lo": 0.0066,
                  "hi": 0.02812,
                  "n": 15,
                  "highlight": false
                },
                {
                  "label": "GPT-6.1 Sol (medium) · Codex CLI",
                  "value": 0.01018,
                  "lo": 0.0054,
                  "hi": 0.02686,
                  "n": 15,
                  "highlight": false
                },
                {
                  "label": "GPT-6.1 Sol (low) · Codex CLI",
                  "value": 0.00769,
                  "lo": 0.00742,
                  "hi": 0.02649,
                  "n": 10,
                  "highlight": false
                }
              ]
            }
          ],
          "note": "Calculation, not a bill: the calls ran on flat subscriptions. Whiskers = cheapest and most expensive call.",
          "sourceIds": [
            "agent-provider-h2h",
            "calc-repricing",
            "price-anthropic",
            "price-openai"
          ]
        },
        {
          "id": "h2h-speed-vs-cost",
          "title": "Speed vs list-price cost",
          "subtitle": "Median total time and median list-price cost per call",
          "kind": "scatter",
          "unit": "seconds",
          "xLabel": "USD per call (list-price calculation)",
          "yLabel": "Median total seconds",
          "series": [
            {
              "name": "Configuration",
              "points": [
                {
                  "label": "Claude Fable 5.1 · Claude Code",
                  "x": 0.00987,
                  "value": 1.94,
                  "n": 15,
                  "highlight": true
                },
                {
                  "label": "Claude Sonnet 5.5 · Claude Code",
                  "x": 0.0036,
                  "value": 2.31,
                  "n": 15,
                  "highlight": true
                },
                {
                  "label": "Claude Opus 5.5 (high) · Claude Code",
                  "x": 0.00694,
                  "value": 2.71,
                  "n": 15,
                  "highlight": true
                },
                {
                  "label": "Claude Opus 5.5 · Claude Code",
                  "x": 0.00688,
                  "value": 2.75,
                  "n": 15,
                  "highlight": true
                },
                {
                  "label": "Claude Opus 5.5 (low) · Claude Code",
                  "x": 0.00688,
                  "value": 2.83,
                  "n": 15,
                  "highlight": true
                },
                {
                  "label": "Claude Haiku 4.5 · Claude Code",
                  "x": 0.00566,
                  "value": 4.43,
                  "n": 15,
                  "highlight": true
                },
                {
                  "label": "GPT-6.1 Sol (high) · Codex CLI",
                  "x": 0.01047,
                  "value": 5.6,
                  "n": 15,
                  "highlight": false
                },
                {
                  "label": "GPT-6.1 Sol (medium) · Codex CLI",
                  "x": 0.01018,
                  "value": 5.65,
                  "n": 15,
                  "highlight": false
                },
                {
                  "label": "GPT-6.1 Sol (low) · Codex CLI",
                  "x": 0.00769,
                  "value": 6.26,
                  "n": 10,
                  "highlight": false
                }
              ]
            }
          ],
          "note": "Lower-left is faster and cheaper. Cost is a calculation from tokens.",
          "sourceIds": [
            "agent-provider-h2h",
            "calc-repricing",
            "price-anthropic",
            "price-openai"
          ]
        },
        {
          "id": "h2h-cost-per-pass",
          "title": "List-price cost per passing answer (calculation)",
          "subtitle": "All calls in a configuration, failures included, divided by its passes",
          "kind": "bar",
          "unit": "usd",
          "yLabel": "USD per passing answer",
          "series": [
            {
              "name": "Cost per pass",
              "points": [
                {
                  "label": "Claude Sonnet 5.5 · Claude Code",
                  "value": 0.00624,
                  "n": 15,
                  "highlight": true
                },
                {
                  "label": "Claude Opus 5.5 (low) · Claude Code",
                  "value": 0.00829,
                  "n": 15,
                  "highlight": true
                },
                {
                  "label": "Claude Haiku 4.5 · Claude Code",
                  "value": 0.00836,
                  "n": 15,
                  "highlight": false
                },
                {
                  "label": "GPT-6.1 Sol (low) · Codex CLI",
                  "value": 0.00998,
                  "n": 10,
                  "highlight": false
                },
                {
                  "label": "Claude Opus 5.5 · Claude Code",
                  "value": 0.01009,
                  "n": 15,
                  "highlight": true
                },
                {
                  "label": "Claude Opus 5.5 (high) · Claude Code",
                  "value": 0.01049,
                  "n": 15,
                  "highlight": true
                },
                {
                  "label": "GPT-6.1 Sol (high) · Codex CLI",
                  "value": 0.01322,
                  "n": 15,
                  "highlight": false
                },
                {
                  "label": "GPT-6.1 Sol (medium) · Codex CLI",
                  "value": 0.01564,
                  "n": 15,
                  "highlight": false
                },
                {
                  "label": "Claude Fable 5.1 · Claude Code",
                  "value": 0.02054,
                  "n": 15,
                  "highlight": true
                }
              ]
            }
          ],
          "note": "Calculation, not a bill: reported tokens × list price; the calls ran on flat subscriptions. A failed call still costs, so a lower pass rate raises the cost per pass. Highlighted bars are on the speed/cost/quality frontier.",
          "sourceIds": [
            "agent-provider-h2h",
            "calc-repricing",
            "price-anthropic",
            "price-openai"
          ]
        },
        {
          "id": "h2h-frontier",
          "title": "Speed, cost and quality frontier",
          "subtitle": "Median seconds against list-price cost per passing answer; pass rate in the note",
          "kind": "scatter",
          "unit": "seconds",
          "xLabel": "USD per passing answer (list-price calculation)",
          "yLabel": "Median total seconds",
          "series": [
            {
              "name": "Claude Code",
              "points": [
                {
                  "label": "Claude Fable 5.1 · Claude Code",
                  "x": 0.02054,
                  "value": 1.94,
                  "n": 15,
                  "highlight": true
                },
                {
                  "label": "Claude Sonnet 5.5 · Claude Code",
                  "x": 0.00624,
                  "value": 2.31,
                  "n": 15,
                  "highlight": true
                },
                {
                  "label": "Claude Opus 5.5 (high) · Claude Code",
                  "x": 0.01049,
                  "value": 2.71,
                  "n": 15,
                  "highlight": true
                },
                {
                  "label": "Claude Opus 5.5 · Claude Code",
                  "x": 0.01009,
                  "value": 2.75,
                  "n": 15,
                  "highlight": true
                },
                {
                  "label": "Claude Opus 5.5 (low) · Claude Code",
                  "x": 0.00829,
                  "value": 2.83,
                  "n": 15,
                  "highlight": true
                },
                {
                  "label": "Claude Haiku 4.5 · Claude Code",
                  "x": 0.00836,
                  "value": 4.43,
                  "n": 15,
                  "highlight": false
                }
              ]
            },
            {
              "name": "Codex CLI",
              "points": [
                {
                  "label": "GPT-6.1 Sol (high) · Codex CLI",
                  "x": 0.01322,
                  "value": 5.6,
                  "n": 15,
                  "highlight": false
                },
                {
                  "label": "GPT-6.1 Sol (medium) · Codex CLI",
                  "x": 0.01564,
                  "value": 5.65,
                  "n": 15,
                  "highlight": false
                },
                {
                  "label": "GPT-6.1 Sol (low) · Codex CLI",
                  "x": 0.00998,
                  "value": 6.26,
                  "n": 10,
                  "highlight": false
                }
              ]
            }
          ],
          "note": "Lower-left is better. Highlighted points are on the frontier: no other configuration is at least as fast, as cheap per pass and as accurate. Frontier: Claude Fable 5.1 · Claude Code; Claude Sonnet 5.5 · Claude Code; Claude Opus 5.5 (high) · Claude Code; Claude Opus 5.5 · Claude Code; Claude Opus 5.5 (low) · Claude Code. Pass rates: Claude Fable 5.1 · Claude Code 15/15; Claude Sonnet 5.5 · Claude Code 12/15; Claude Opus 5.5 (high) · Claude Code 15/15; Claude Opus 5.5 · Claude Code 15/15; Claude Opus 5.5 (low) · Claude Code 15/15; Claude Haiku 4.5 · Claude Code 15/15; GPT-6.1 Sol (high) · Codex CLI 15/15; GPT-6.1 Sol (medium) · Codex CLI 15/15; GPT-6.1 Sol (low) · Codex CLI 10/10. Costs are calculations from tokens; medians come from small samples.",
          "sourceIds": [
            "agent-provider-h2h",
            "calc-repricing",
            "price-anthropic",
            "price-openai"
          ]
        }
      ],
      "tables": [
        {
          "id": "h2h-pass-matrix",
          "title": "Pass matrix: configuration × task",
          "columns": [
            {
              "key": "config",
              "label": "Configuration",
              "unit": "text"
            },
            {
              "key": "c0",
              "label": "Fix a buggy median function",
              "unit": "text"
            },
            {
              "key": "c1",
              "label": "Extract invoice fields to JSON",
              "unit": "text"
            },
            {
              "key": "c2",
              "label": "Multi-step shift arithmetic",
              "unit": "text"
            },
            {
              "key": "c3",
              "label": "Refactor recursion to iteration",
              "unit": "text"
            },
            {
              "key": "c4",
              "label": "Classify six support tickets",
              "unit": "text"
            },
            {
              "key": "medianTotal",
              "label": "Median total (s)",
              "unit": "seconds"
            }
          ],
          "rows": [
            {
              "config": "Claude Fable 5.1 · Claude Code",
              "medianTotal": 1.94,
              "c0": "3/3",
              "c1": "3/3",
              "c2": "3/3",
              "c3": "3/3",
              "c4": "3/3"
            },
            {
              "config": "Claude Sonnet 5.5 · Claude Code",
              "medianTotal": 2.31,
              "c0": "3/3",
              "c1": "3/3",
              "c2": "0/3",
              "c3": "3/3",
              "c4": "3/3"
            },
            {
              "config": "Claude Opus 5.5 (high) · Claude Code",
              "medianTotal": 2.71,
              "c0": "3/3",
              "c1": "3/3",
              "c2": "3/3",
              "c3": "3/3",
              "c4": "3/3"
            },
            {
              "config": "Claude Opus 5.5 · Claude Code",
              "medianTotal": 2.75,
              "c0": "3/3",
              "c1": "3/3",
              "c2": "3/3",
              "c3": "3/3",
              "c4": "3/3"
            },
            {
              "config": "Claude Opus 5.5 (low) · Claude Code",
              "medianTotal": 2.83,
              "c0": "3/3",
              "c1": "3/3",
              "c2": "3/3",
              "c3": "3/3",
              "c4": "3/3"
            },
            {
              "config": "Claude Haiku 4.5 · Claude Code",
              "medianTotal": 4.43,
              "c0": "3/3",
              "c1": "3/3",
              "c2": "3/3",
              "c3": "3/3",
              "c4": "3/3"
            },
            {
              "config": "GPT-6.1 Sol (high) · Codex CLI",
              "medianTotal": 5.6,
              "c0": "3/3",
              "c1": "3/3",
              "c2": "3/3",
              "c3": "3/3",
              "c4": "3/3"
            },
            {
              "config": "GPT-6.1 Sol (medium) · Codex CLI",
              "medianTotal": 5.65,
              "c0": "3/3",
              "c1": "3/3",
              "c2": "3/3",
              "c3": "3/3",
              "c4": "3/3"
            },
            {
              "config": "GPT-6.1 Sol (low) · Codex CLI",
              "medianTotal": 6.26,
              "c0": "2/2",
              "c1": "2/2",
              "c2": "2/2",
              "c3": "2/2",
              "c4": "2/2"
            }
          ]
        }
      ],
      "related": [
        "hard-model-head-to-head"
      ]
    },
    {
      "slug": "hard-model-head-to-head",
      "title": "Haiku vs Sonnet vs Opus vs Fable vs GPT-6.1 Sol on 8 hard tasks",
      "seoTitle": "Hard tasks: Haiku vs Sonnet vs Opus vs Fable vs GPT-6.1 Sol",
      "description": "152 calls on 8 hard tasks with strict validators. Pass rate with 95% intervals, format misses, speed, tokens and cost per pass.",
      "question": "On 8 hard tasks with deterministic validators, does pass rate separate the Claude Code models and GPT-6.1 Sol through the Codex CLI, and what do speed, tokens and cost per pass add?",
      "answer": "139 of 152 calls that reached a model passed strictly (91%). 6 of 7 configurations passed every call: Claude Sonnet 5.5 · Claude Code, Claude Opus 5.5 · Claude Code, Claude Opus 5.5 (high) · Claude Code, Claude Fable 5.1 · Claude Code (24/24 each, 95% interval 86% to 100%) and GPT-6.1 Sol (medium) · Codex CLI, GPT-6.1 Sol (high) · Codex CLI (16/16 each, 95% interval 81% to 100%), so the hard set still has a ceiling for these models and pass rate does not separate them. Claude Haiku 4.5 · Claude Code passed 11/24 strictly (46%, 95% interval 28% to 65%). 5 more replies had the right answer in the wrong format (for example inside a code fence), so 16/24 on a lenient reading (47% to 82%); 8 replies were wrong. It passed none of the tasks “Predict JavaScript event-loop output order”, “Solve a multi-constraint room schedule” and “Write a SQLite reporting query (fan-out, ties, boundaries)”. Median total time per call was Sonnet 7.7 s, Opus 9.2 s, Opus (high) 11.0 s, GPT-6.1 Sol (medium) 13.1 s, Fable 16.1 s, GPT-6.1 Sol (high) 18.1 s, Haiku 39.0 s. The fastest and slowest single calls of every configuration overlap with every other, so these medians describe this run; they are not a tested ranking. At list price (a calculation; the calls ran on a subscription), the lowest cost per strict pass was Claude Sonnet 5.5 · Claude Code at $0.0143; the quality-vs-cost frontier is Claude Sonnet 5.5 · Claude Code. 30 earlier attempts were blocked before any model call (Codex CLI: the CLI reported no signed-in account); they are reported, not scored, and that route ran in a later batch.",
      "date": "2026-10-06",
      "updated": "2026-10-06",
      "tags": [
        "head-to-head",
        "hard-tasks",
        "claude-haiku",
        "claude-sonnet",
        "claude-opus",
        "claude-fable",
        "gpt-6-1-sol",
        "codex-cli",
        "format-misses",
        "latency"
      ],
      "method": [
        "Protocol declared before the first call. A follow-up to the five-task head-to-head (/benchmarks/model-head-to-head), where pass rate hit a ceiling.",
        "8 hard tasks, each with a deterministic validator that runs in a sandbox without network: Fix an interval-merge function (off-by-one and edge cases); Fix a time-zone day-length function (DST); Write a CSV parser (quoted newlines, strict errors); Predict JavaScript event-loop output order; Solve a multi-constraint room schedule; Write a strict SemVer 2.0.0 regex; Refactor to remove duplication, keep 20 tests green; Write a SQLite reporting query (fan-out, ties, boundaries).",
        "Controls before inference: 8/8 reference answers pass, 26/26 plausible wrong answers fail, and 8/8 references wrapped in a fence or prose are flagged as format misses.",
        "Configurations that reached a model: Sonnet, Opus, Opus (high), GPT-6.1 Sol (medium), Fable, GPT-6.1 Sol (high), Haiku. Claude Code: n = 24 per configuration (3 repetitions per task); Codex CLI: n = 16 per configuration (2 repetitions per task).",
        "Strict pass: the whole reply, trimmed, passes the validator. Every prompt states the output format and says no code fence and no other text. This is stricter than the five-task study, which removed one wrapping fence.",
        "Format miss: the strict check failed, but a lenient extractor (fenced block, outer JSON, first code line, one output line) finds an answer that passes the same validator. Reported apart from wrong answers, never as a pass.",
        "Isolation: fresh empty working folder, tools off, no MCP servers, no session persistence, one turn, 300 s timeout per call. One call at a time per account.",
        "Attempts: Claude Code: 120 attempts, 120 reached a model, 0 blocked; Codex CLI: 62 attempts, 32 reached a model, 30 blocked. Nothing was trimmed or retried, and no run hit a usage or rate limit.",
        "One Codex batch was resumed after its process ended (declared in the protocol before the resume): the resume skipped every task, repetition and effort the batch file already held, so no recorded call was repeated or replaced.",
        "Cost per strict pass: list price × reported tokens for every call in the configuration (cache reads and writes priced as in the five-task study), divided by its strict passes. A calculation."
      ],
      "caveats": [
        "6 configurations passed every call, so the hard set still has a ceiling for them: a perfect 24/24 has a 95% interval of 86% to 100%; a perfect 16/16 has a 95% interval of 81% to 100%. Among them, only the latency, token and cost medians differ, and their per-call time ranges overlap.",
        "Claude Code: n = 24 per configuration (3 repetitions per task); Codex CLI: n = 16 per configuration (2 repetitions per task). Per-task cells have only 2 to 3 calls.",
        "Claude Code and Codex CLI rows pair a CLI with a model, and each CLI adds its own system prompt and start-up time. A Claude-vs-GPT row compares the route + model pairs, not the models alone.",
        "The Claude and Codex batches ran on different days on the same host, one call at a time per account. Each route used its own subscription.",
        "Strict format rules decide part of the result: a reply in a code fence fails. The lenient reading is shown next to it so the two can be told apart.",
        "CLI timings include CLI start-up and the CLI’s own system prompt. One host and network; the counted Claude and Codex batches ran hours apart. Host load was not controlled.",
        "Default effort means the effort flag was not passed; the CLI chose. Haiku reported a median 4,556 reasoning tokens per call, which explains much of its extra time and output.",
        "List-price costs are calculations; the calls used a flat subscription."
      ],
      "sourceIds": [
        "agent-provider-h2h-hard",
        "calc-repricing",
        "price-anthropic",
        "price-openai"
      ],
      "stats": [
        {
          "id": "hard-h2h-pass-all",
          "label": "Calls that passed strictly (hard set)",
          "value": 0.9145,
          "unit": "rate",
          "display": "91% (139/152)",
          "n": 152,
          "ci": [
            0.8592,
            0.9493
          ]
        },
        {
          "id": "hard-h2h-correct-all",
          "label": "Calls with a correct answer, format misses included (lenient reading)",
          "value": 0.9474,
          "unit": "rate",
          "display": "95% (144/152)",
          "n": 152,
          "ci": [
            0.8996,
            0.9731
          ]
        },
        {
          "id": "hard-h2h-format-misses",
          "label": "Non-passes that were format misses, not wrong answers",
          "value": 5,
          "unit": "count",
          "display": "5 of 13 non-passes (8 wrong answers)",
          "n": 13
        },
        {
          "id": "hard-h2h-perfect-configs",
          "label": "Configurations that passed every call",
          "value": 6,
          "unit": "count",
          "display": "6 of 7 (4 at 24/24, 2 at 16/16)",
          "n": 7
        },
        {
          "id": "hard-h2h-fastest-perfect",
          "label": "Lowest observed median time among configurations that passed every call (separate batches)",
          "value": 7.75,
          "unit": "seconds",
          "display": "Claude Sonnet 5.5 · Claude Code: 7.7 s",
          "n": 24,
          "note": "The counted Claude and Codex batches ran hours apart on one host and network. Host load was not controlled; this does not isolate model speed."
        },
        {
          "id": "hard-h2h-cheapest-per-pass",
          "label": "Lowest list-price cost per strict pass (calculation)",
          "value": 0.01435,
          "unit": "usd",
          "display": "Claude Sonnet 5.5 · Claude Code: $0.0143",
          "n": 24
        },
        {
          "id": "hard-h2h-blocked",
          "label": "Attempts blocked before any model call (not scored)",
          "value": 30,
          "unit": "count",
          "display": "30 (Codex CLI; 0 model calls)",
          "n": 182
        }
      ],
      "charts": [
        {
          "id": "hard-h2h-pass-rate",
          "title": "Pass rate on eight hard tasks",
          "subtitle": "Strict: the reply passes as given. Lenient: a correct answer in the wrong format also counts",
          "kind": "dot-range",
          "unit": "rate",
          "yLabel": "Passed",
          "series": [
            {
              "name": "Strict pass",
              "points": [
                {
                  "label": "Claude Sonnet 5.5 · Claude Code",
                  "value": 1,
                  "lo": 0.862,
                  "hi": 1,
                  "n": 24
                },
                {
                  "label": "Claude Opus 5.5 · Claude Code",
                  "value": 1,
                  "lo": 0.862,
                  "hi": 1,
                  "n": 24
                },
                {
                  "label": "Claude Opus 5.5 (high) · Claude Code",
                  "value": 1,
                  "lo": 0.862,
                  "hi": 1,
                  "n": 24
                },
                {
                  "label": "GPT-6.1 Sol (medium) · Codex CLI",
                  "value": 1,
                  "lo": 0.8064,
                  "hi": 1,
                  "n": 16
                },
                {
                  "label": "Claude Fable 5.1 · Claude Code",
                  "value": 1,
                  "lo": 0.862,
                  "hi": 1,
                  "n": 24
                },
                {
                  "label": "GPT-6.1 Sol (high) · Codex CLI",
                  "value": 1,
                  "lo": 0.8064,
                  "hi": 1,
                  "n": 16
                },
                {
                  "label": "Claude Haiku 4.5 · Claude Code",
                  "value": 0.4583,
                  "lo": 0.2789,
                  "hi": 0.6493,
                  "n": 24
                }
              ]
            },
            {
              "name": "Lenient (format misses counted)",
              "points": [
                {
                  "label": "Claude Sonnet 5.5 · Claude Code",
                  "value": 1,
                  "lo": 0.862,
                  "hi": 1,
                  "n": 24
                },
                {
                  "label": "Claude Opus 5.5 · Claude Code",
                  "value": 1,
                  "lo": 0.862,
                  "hi": 1,
                  "n": 24
                },
                {
                  "label": "Claude Opus 5.5 (high) · Claude Code",
                  "value": 1,
                  "lo": 0.862,
                  "hi": 1,
                  "n": 24
                },
                {
                  "label": "GPT-6.1 Sol (medium) · Codex CLI",
                  "value": 1,
                  "lo": 0.8064,
                  "hi": 1,
                  "n": 16
                },
                {
                  "label": "Claude Fable 5.1 · Claude Code",
                  "value": 1,
                  "lo": 0.862,
                  "hi": 1,
                  "n": 24
                },
                {
                  "label": "GPT-6.1 Sol (high) · Codex CLI",
                  "value": 1,
                  "lo": 0.8064,
                  "hi": 1,
                  "n": 16
                },
                {
                  "label": "Claude Haiku 4.5 · Claude Code",
                  "value": 0.6667,
                  "lo": 0.4671,
                  "hi": 0.8203,
                  "n": 24
                }
              ]
            }
          ],
          "note": "Whiskers are 95% Wilson intervals. The strict pass is the result; the lenient reading is shown so a reader can see how much is format and how much is a wrong answer. A format miss never counts as a pass.",
          "sourceIds": [
            "agent-provider-h2h-hard"
          ]
        },
        {
          "id": "hard-h2h-outcomes",
          "title": "What happened on every call",
          "subtitle": "Counts per configuration: strict passes, format misses and wrong answers",
          "kind": "stacked-bar",
          "unit": "count",
          "yLabel": "Calls",
          "series": [
            {
              "name": "Strict pass",
              "points": [
                {
                  "label": "Claude Sonnet 5.5 · Claude Code",
                  "value": 24,
                  "n": 24
                },
                {
                  "label": "Claude Opus 5.5 · Claude Code",
                  "value": 24,
                  "n": 24
                },
                {
                  "label": "Claude Opus 5.5 (high) · Claude Code",
                  "value": 24,
                  "n": 24
                },
                {
                  "label": "GPT-6.1 Sol (medium) · Codex CLI",
                  "value": 16,
                  "n": 16
                },
                {
                  "label": "Claude Fable 5.1 · Claude Code",
                  "value": 24,
                  "n": 24
                },
                {
                  "label": "GPT-6.1 Sol (high) · Codex CLI",
                  "value": 16,
                  "n": 16
                },
                {
                  "label": "Claude Haiku 4.5 · Claude Code",
                  "value": 11,
                  "n": 24
                }
              ]
            },
            {
              "name": "Format miss (correct answer, wrong format)",
              "points": [
                {
                  "label": "Claude Sonnet 5.5 · Claude Code",
                  "value": 0,
                  "n": 24
                },
                {
                  "label": "Claude Opus 5.5 · Claude Code",
                  "value": 0,
                  "n": 24
                },
                {
                  "label": "Claude Opus 5.5 (high) · Claude Code",
                  "value": 0,
                  "n": 24
                },
                {
                  "label": "GPT-6.1 Sol (medium) · Codex CLI",
                  "value": 0,
                  "n": 16
                },
                {
                  "label": "Claude Fable 5.1 · Claude Code",
                  "value": 0,
                  "n": 24
                },
                {
                  "label": "GPT-6.1 Sol (high) · Codex CLI",
                  "value": 0,
                  "n": 16
                },
                {
                  "label": "Claude Haiku 4.5 · Claude Code",
                  "value": 5,
                  "n": 24
                }
              ]
            },
            {
              "name": "Wrong answer",
              "points": [
                {
                  "label": "Claude Sonnet 5.5 · Claude Code",
                  "value": 0,
                  "n": 24
                },
                {
                  "label": "Claude Opus 5.5 · Claude Code",
                  "value": 0,
                  "n": 24
                },
                {
                  "label": "Claude Opus 5.5 (high) · Claude Code",
                  "value": 0,
                  "n": 24
                },
                {
                  "label": "GPT-6.1 Sol (medium) · Codex CLI",
                  "value": 0,
                  "n": 16
                },
                {
                  "label": "Claude Fable 5.1 · Claude Code",
                  "value": 0,
                  "n": 24
                },
                {
                  "label": "GPT-6.1 Sol (high) · Codex CLI",
                  "value": 0,
                  "n": 16
                },
                {
                  "label": "Claude Haiku 4.5 · Claude Code",
                  "value": 8,
                  "n": 24
                }
              ]
            }
          ],
          "note": "A format miss is a reply that the strict validator rejected (for example, wrapped in a code fence or in prose although the prompt said not to) whose extracted answer passes the same validator. It is not a pass.",
          "sourceIds": [
            "agent-provider-h2h-hard"
          ]
        },
        {
          "id": "hard-h2h-total-latency",
          "title": "Total time per call on hard tasks (separate batches)",
          "subtitle": "Median per configuration; whiskers = fastest and slowest call",
          "kind": "dot-range",
          "unit": "seconds",
          "yLabel": "Seconds",
          "series": [
            {
              "name": "Total time per call on hard tasks (separate batches)",
              "points": [
                {
                  "label": "Claude Sonnet 5.5 · Claude Code",
                  "value": 7.75,
                  "lo": 2.26,
                  "hi": 34.79,
                  "n": 24,
                  "highlight": true
                },
                {
                  "label": "Claude Opus 5.5 · Claude Code",
                  "value": 9.18,
                  "lo": 4.24,
                  "hi": 27.21,
                  "n": 24,
                  "highlight": true
                },
                {
                  "label": "Claude Opus 5.5 (high) · Claude Code",
                  "value": 11.03,
                  "lo": 3.63,
                  "hi": 63,
                  "n": 24,
                  "highlight": true
                },
                {
                  "label": "GPT-6.1 Sol (medium) · Codex CLI",
                  "value": 13.11,
                  "lo": 8.54,
                  "hi": 61.6,
                  "n": 16,
                  "highlight": true
                },
                {
                  "label": "Claude Fable 5.1 · Claude Code",
                  "value": 16.13,
                  "lo": 4.46,
                  "hi": 90,
                  "n": 24,
                  "highlight": true
                },
                {
                  "label": "GPT-6.1 Sol (high) · Codex CLI",
                  "value": 18.12,
                  "lo": 11.67,
                  "hi": 92.21,
                  "n": 16,
                  "highlight": true
                },
                {
                  "label": "Claude Haiku 4.5 · Claude Code",
                  "value": 39.01,
                  "lo": 15.27,
                  "hi": 75.13,
                  "n": 24,
                  "highlight": false
                }
              ]
            }
          ],
          "note": "One host and network; the counted Claude and Codex batches ran hours apart. Host load was not controlled. Whiskers are a range, not a confidence interval. Highlighted: configurations that passed every call.",
          "sourceIds": [
            "agent-provider-h2h-hard"
          ]
        },
        {
          "id": "hard-h2h-first-useful-latency",
          "title": "Time to first useful output on hard tasks",
          "subtitle": "Median per configuration; whiskers = fastest and slowest call",
          "kind": "dot-range",
          "unit": "seconds",
          "yLabel": "Seconds",
          "series": [
            {
              "name": "Time to first useful output on hard tasks",
              "points": [
                {
                  "label": "Claude Sonnet 5.5 · Claude Code",
                  "value": 5.95,
                  "lo": 0.86,
                  "hi": 30.57,
                  "n": 24,
                  "highlight": true
                },
                {
                  "label": "Claude Opus 5.5 · Claude Code",
                  "value": 6.78,
                  "lo": 2.39,
                  "hi": 21.77,
                  "n": 24,
                  "highlight": true
                },
                {
                  "label": "Claude Opus 5.5 (high) · Claude Code",
                  "value": 7.13,
                  "lo": 2.15,
                  "hi": 56.23,
                  "n": 24,
                  "highlight": true
                },
                {
                  "label": "GPT-6.1 Sol (medium) · Codex CLI",
                  "value": 10.23,
                  "lo": 6.09,
                  "hi": 40.41,
                  "n": 16,
                  "highlight": true
                },
                {
                  "label": "Claude Fable 5.1 · Claude Code",
                  "value": 11.63,
                  "lo": 2,
                  "hi": 85.33,
                  "n": 24,
                  "highlight": true
                },
                {
                  "label": "GPT-6.1 Sol (high) · Codex CLI",
                  "value": 12.69,
                  "lo": 8.93,
                  "hi": 75.91,
                  "n": 16,
                  "highlight": true
                },
                {
                  "label": "Claude Haiku 4.5 · Claude Code",
                  "value": 35.54,
                  "lo": 12.88,
                  "hi": 70.31,
                  "n": 24,
                  "highlight": false
                }
              ]
            }
          ],
          "note": "One host and network; the counted Claude and Codex batches ran hours apart. Host load was not controlled. Whiskers are a range, not a confidence interval. Highlighted: configurations that passed every call.",
          "sourceIds": [
            "agent-provider-h2h-hard"
          ]
        },
        {
          "id": "hard-h2h-output-tokens",
          "title": "Output tokens per call on hard tasks",
          "subtitle": "Median per configuration; reasoning tokens as the CLI reports them",
          "kind": "grouped-bar",
          "unit": "tokens",
          "yLabel": "Tokens",
          "series": [
            {
              "name": "Output tokens",
              "points": [
                {
                  "label": "Claude Sonnet 5.5 · Claude Code",
                  "value": 1050,
                  "n": 24
                },
                {
                  "label": "Claude Opus 5.5 · Claude Code",
                  "value": 945,
                  "n": 24
                },
                {
                  "label": "Claude Opus 5.5 (high) · Claude Code",
                  "value": 1052,
                  "n": 24
                },
                {
                  "label": "GPT-6.1 Sol (medium) · Codex CLI",
                  "value": 335,
                  "n": 16
                },
                {
                  "label": "Claude Fable 5.1 · Claude Code",
                  "value": 1366,
                  "n": 24
                },
                {
                  "label": "GPT-6.1 Sol (high) · Codex CLI",
                  "value": 436,
                  "n": 16
                },
                {
                  "label": "Claude Haiku 4.5 · Claude Code",
                  "value": 5064,
                  "n": 24
                }
              ]
            },
            {
              "name": "Reasoning tokens",
              "points": [
                {
                  "label": "Claude Sonnet 5.5 · Claude Code",
                  "value": 585,
                  "n": 24
                },
                {
                  "label": "Claude Opus 5.5 · Claude Code",
                  "value": 529,
                  "n": 24
                },
                {
                  "label": "Claude Opus 5.5 (high) · Claude Code",
                  "value": 614,
                  "n": 24
                },
                {
                  "label": "GPT-6.1 Sol (medium) · Codex CLI",
                  "value": 150,
                  "n": 16
                },
                {
                  "label": "Claude Fable 5.1 · Claude Code",
                  "value": 889,
                  "n": 24
                },
                {
                  "label": "GPT-6.1 Sol (high) · Codex CLI",
                  "value": 225,
                  "n": 16
                },
                {
                  "label": "Claude Haiku 4.5 · Claude Code",
                  "value": 4556,
                  "n": 24
                }
              ]
            }
          ],
          "note": "Reasoning tokens are part of the output tokens where the CLI reports them. Their content is never captured. More tokens is not better or worse by itself.",
          "sourceIds": [
            "agent-provider-h2h-hard"
          ]
        },
        {
          "id": "hard-h2h-cost-per-pass",
          "title": "List-price cost per strict pass on hard tasks (calculation)",
          "subtitle": "All calls in a configuration, failures and format misses included, divided by its strict passes",
          "kind": "bar",
          "unit": "usd",
          "yLabel": "USD per strict pass",
          "series": [
            {
              "name": "Cost per strict pass",
              "points": [
                {
                  "label": "Claude Sonnet 5.5 · Claude Code",
                  "value": 0.01435,
                  "n": 24,
                  "highlight": true
                },
                {
                  "label": "GPT-6.1 Sol (high) · Codex CLI",
                  "value": 0.01514,
                  "n": 16,
                  "highlight": false
                },
                {
                  "label": "GPT-6.1 Sol (medium) · Codex CLI",
                  "value": 0.02564,
                  "n": 16,
                  "highlight": false
                },
                {
                  "label": "Claude Opus 5.5 · Claude Code",
                  "value": 0.02824,
                  "n": 24,
                  "highlight": false
                },
                {
                  "label": "Claude Opus 5.5 (high) · Claude Code",
                  "value": 0.03337,
                  "n": 24,
                  "highlight": false
                },
                {
                  "label": "Claude Haiku 4.5 · Claude Code",
                  "value": 0.0672,
                  "n": 24,
                  "highlight": false
                },
                {
                  "label": "Claude Fable 5.1 · Claude Code",
                  "value": 0.09331,
                  "n": 24,
                  "highlight": false
                }
              ]
            }
          ],
          "note": "Calculation, not a bill: reported tokens × list price; the calls ran on a flat subscription. A failed call still costs, so a lower pass rate raises the cost per pass. Highlighted bars are on the quality-vs-cost frontier.",
          "sourceIds": [
            "agent-provider-h2h-hard",
            "calc-repricing",
            "price-anthropic",
            "price-openai"
          ]
        },
        {
          "id": "hard-h2h-frontier",
          "title": "Quality vs cost frontier on hard tasks",
          "subtitle": "Strict pass rate against list-price cost per strict pass",
          "kind": "scatter",
          "unit": "rate",
          "xLabel": "USD per strict pass (list-price calculation)",
          "yLabel": "Strict pass rate",
          "series": [
            {
              "name": "Claude Code",
              "points": [
                {
                  "label": "Claude Sonnet 5.5 · Claude Code",
                  "x": 0.01435,
                  "value": 1,
                  "n": 24,
                  "highlight": true
                },
                {
                  "label": "Claude Opus 5.5 · Claude Code",
                  "x": 0.02824,
                  "value": 1,
                  "n": 24,
                  "highlight": false
                },
                {
                  "label": "Claude Opus 5.5 (high) · Claude Code",
                  "x": 0.03337,
                  "value": 1,
                  "n": 24,
                  "highlight": false
                },
                {
                  "label": "Claude Fable 5.1 · Claude Code",
                  "x": 0.09331,
                  "value": 1,
                  "n": 24,
                  "highlight": false
                },
                {
                  "label": "Claude Haiku 4.5 · Claude Code",
                  "x": 0.0672,
                  "value": 0.4583,
                  "n": 24,
                  "highlight": false
                }
              ]
            },
            {
              "name": "Codex CLI",
              "points": [
                {
                  "label": "GPT-6.1 Sol (medium) · Codex CLI",
                  "x": 0.02564,
                  "value": 1,
                  "n": 16,
                  "highlight": false
                },
                {
                  "label": "GPT-6.1 Sol (high) · Codex CLI",
                  "x": 0.01514,
                  "value": 1,
                  "n": 16,
                  "highlight": false
                }
              ]
            }
          ],
          "note": "Upper-left is better. Highlighted points are on the frontier: no other configuration passes at least as often for at most the same cost per pass. Frontier: Claude Sonnet 5.5 · Claude Code. Costs are calculations from tokens. Pass rates with their 95% intervals are in the pass-rate chart.",
          "sourceIds": [
            "agent-provider-h2h-hard",
            "calc-repricing",
            "price-anthropic",
            "price-openai"
          ]
        }
      ],
      "tables": [
        {
          "id": "hard-h2h-pass-matrix",
          "title": "Pass matrix on hard tasks: configuration × task (strict passes)",
          "columns": [
            {
              "key": "config",
              "label": "Configuration",
              "unit": "text"
            },
            {
              "key": "c0",
              "label": "Fix an interval-merge function (off-by-one and edge cases)",
              "unit": "text"
            },
            {
              "key": "c1",
              "label": "Fix a time-zone day-length function (DST)",
              "unit": "text"
            },
            {
              "key": "c2",
              "label": "Write a CSV parser (quoted newlines, strict errors)",
              "unit": "text"
            },
            {
              "key": "c3",
              "label": "Predict JavaScript event-loop output order",
              "unit": "text"
            },
            {
              "key": "c4",
              "label": "Solve a multi-constraint room schedule",
              "unit": "text"
            },
            {
              "key": "c5",
              "label": "Write a strict SemVer 2.0.0 regex",
              "unit": "text"
            },
            {
              "key": "c6",
              "label": "Refactor to remove duplication, keep 20 tests green",
              "unit": "text"
            },
            {
              "key": "c7",
              "label": "Write a SQLite reporting query (fan-out, ties, boundaries)",
              "unit": "text"
            },
            {
              "key": "strict",
              "label": "Strict total",
              "unit": "text"
            },
            {
              "key": "formatMisses",
              "label": "Format misses",
              "unit": "count"
            },
            {
              "key": "wrong",
              "label": "Wrong answers",
              "unit": "count"
            },
            {
              "key": "medianTotal",
              "label": "Median total (s)",
              "unit": "seconds"
            }
          ],
          "rows": [
            {
              "config": "Claude Sonnet 5.5 · Claude Code",
              "strict": "24/24",
              "formatMisses": 0,
              "wrong": 0,
              "medianTotal": 7.75,
              "c0": "3/3",
              "c1": "3/3",
              "c2": "3/3",
              "c3": "3/3",
              "c4": "3/3",
              "c5": "3/3",
              "c6": "3/3",
              "c7": "3/3"
            },
            {
              "config": "Claude Opus 5.5 · Claude Code",
              "strict": "24/24",
              "formatMisses": 0,
              "wrong": 0,
              "medianTotal": 9.18,
              "c0": "3/3",
              "c1": "3/3",
              "c2": "3/3",
              "c3": "3/3",
              "c4": "3/3",
              "c5": "3/3",
              "c6": "3/3",
              "c7": "3/3"
            },
            {
              "config": "Claude Opus 5.5 (high) · Claude Code",
              "strict": "24/24",
              "formatMisses": 0,
              "wrong": 0,
              "medianTotal": 11.03,
              "c0": "3/3",
              "c1": "3/3",
              "c2": "3/3",
              "c3": "3/3",
              "c4": "3/3",
              "c5": "3/3",
              "c6": "3/3",
              "c7": "3/3"
            },
            {
              "config": "GPT-6.1 Sol (medium) · Codex CLI",
              "strict": "16/16",
              "formatMisses": 0,
              "wrong": 0,
              "medianTotal": 13.11,
              "c0": "2/2",
              "c1": "2/2",
              "c2": "2/2",
              "c3": "2/2",
              "c4": "2/2",
              "c5": "2/2",
              "c6": "2/2",
              "c7": "2/2"
            },
            {
              "config": "Claude Fable 5.1 · Claude Code",
              "strict": "24/24",
              "formatMisses": 0,
              "wrong": 0,
              "medianTotal": 16.13,
              "c0": "3/3",
              "c1": "3/3",
              "c2": "3/3",
              "c3": "3/3",
              "c4": "3/3",
              "c5": "3/3",
              "c6": "3/3",
              "c7": "3/3"
            },
            {
              "config": "GPT-6.1 Sol (high) · Codex CLI",
              "strict": "16/16",
              "formatMisses": 0,
              "wrong": 0,
              "medianTotal": 18.12,
              "c0": "2/2",
              "c1": "2/2",
              "c2": "2/2",
              "c3": "2/2",
              "c4": "2/2",
              "c5": "2/2",
              "c6": "2/2",
              "c7": "2/2"
            },
            {
              "config": "Claude Haiku 4.5 · Claude Code",
              "strict": "11/24",
              "formatMisses": 5,
              "wrong": 8,
              "medianTotal": 39.01,
              "c0": "3/3",
              "c1": "1/3",
              "c2": "2/3 (+1 format miss)",
              "c3": "0/3",
              "c4": "0/3 (+2 format misses)",
              "c5": "3/3",
              "c6": "2/3",
              "c7": "0/3 (+2 format misses)"
            }
          ]
        },
        {
          "id": "hard-h2h-controls",
          "title": "Validator controls run before the first model call",
          "columns": [
            {
              "key": "task",
              "label": "Task",
              "unit": "text"
            },
            {
              "key": "reference",
              "label": "Reference answer",
              "unit": "text"
            },
            {
              "key": "checks",
              "label": "Checks",
              "unit": "count"
            },
            {
              "key": "wrong",
              "label": "Plausible wrong answers rejected",
              "unit": "text"
            },
            {
              "key": "wrapped",
              "label": "Wrapped reference flagged as format miss",
              "unit": "text"
            }
          ],
          "rows": [
            {
              "task": "Fix an interval-merge function (off-by-one and edge cases)",
              "reference": "passes",
              "checks": 18,
              "wrong": "5/5",
              "wrapped": "yes"
            },
            {
              "task": "Fix a time-zone day-length function (DST)",
              "reference": "passes",
              "checks": 36,
              "wrong": "3/3",
              "wrapped": "yes"
            },
            {
              "task": "Write a CSV parser (quoted newlines, strict errors)",
              "reference": "passes",
              "checks": 22,
              "wrong": "3/3",
              "wrapped": "yes"
            },
            {
              "task": "Predict JavaScript event-loop output order",
              "reference": "passes",
              "checks": 1,
              "wrong": "3/3",
              "wrapped": "yes"
            },
            {
              "task": "Solve a multi-constraint room schedule",
              "reference": "passes",
              "checks": 14,
              "wrong": "3/3",
              "wrapped": "yes"
            },
            {
              "task": "Write a strict SemVer 2.0.0 regex",
              "reference": "passes",
              "checks": 30,
              "wrong": "3/3",
              "wrapped": "yes"
            },
            {
              "task": "Refactor to remove duplication, keep 20 tests green",
              "reference": "passes",
              "checks": 25,
              "wrong": "2/2",
              "wrapped": "yes"
            },
            {
              "task": "Write a SQLite reporting query (fan-out, ties, boundaries)",
              "reference": "passes",
              "checks": 8,
              "wrong": "4/4",
              "wrapped": "yes"
            }
          ]
        }
      ],
      "related": [
        "model-head-to-head"
      ]
    },
    {
      "slug": "coding-agents-head-to-head",
      "title": "Claude Code (Sonnet 5.5, Opus 5.5) vs Codex CLI on 6 hidden-test coding tasks",
      "seoTitle": "Claude Code vs Codex CLI: 6 coding tasks, hidden tests",
      "description": "36 graded sessions: Claude Code with Sonnet 5.5 and Opus 5.5, Codex CLI with GPT-6.1 Sol. All passed every hidden test; time, tool calls and diffs differ.",
      "question": "On small real repository tasks graded by hidden tests, how do coding-agent CLIs compare when they run with their normal file and shell tools?",
      "answer": "All 3 agents passed every hidden check in every session (12/12, 12/12, 12/12; 95% Wilson 76–100% each), so this task set cannot separate them on quality. Sonnet 5.5 in Claude Code was fastest (median 23.1 s; its sessions took 18.7 s to 44.5 s), Opus 5.5 in Claude Code took a median 56.9 s, GPT-6.1 Sol in Codex CLI took a median 113.4 s; the run ranges of Sonnet 5.5 in Claude Code and GPT-6.1 Sol in Codex CLI do not overlap. Codex made more tool calls (median 12.5 vs 7.5 and 7.5) and larger diffs (median 94 lines vs 45 and 77.5, mostly added tests), and every Codex session also followed the tester’s global AGENTS.md (10 of 12 wrote a work log nobody asked for), so its time and diff include extra work. Opus used 1.9× the output tokens of Sonnet in the same CLI (medians). Gemini CLI was not run: it needed a browser login.",
      "date": "2026-10-06",
      "updated": "2026-10-06",
      "tags": [
        "claude-code",
        "codex-cli",
        "coding-agents",
        "hidden-tests",
        "head-to-head"
      ],
      "method": [
        "Six small Node.js repositories: a pagination bug across two files, a CLI flag with exact error text, a refactor that must keep every quirk, an LRU cache to a written spec, a queue race and strict TypeScript types. Each has 1 to 4 visible tests; the hidden checks live outside the repository and run only after the session ends.",
        "Controls before the first session (Node v25.2.1): every base repository fails its hidden checks and every reference patch passes all of them.",
        "Claude Code 2.1.286 headless with the sonnet and opus models at the default effort; tools Bash, Read, Edit, Write, Glob, Grep; no web tools, MCP servers or sub-agents; the OS sandbox on, writes limited to the repository, no network; setting sources off. Codex CLI exec with GPT-6.1 Sol at medium effort, workspace-write sandbox, no network, user config ignored (its version was not recorded).",
        "6 tasks × 2 repetitions × 3 agents = 36 sessions with a 10-minute timeout; nothing retried. The Claude Code and Codex CLI lanes ran at the same time on one machine with different accounts.",
        "Pass = every hidden check passes; no partial credit. Recorded per session: wall time, tool calls, tokens as reported, files touched, lines changed against the base commit, commits and edits outside the repository.",
        "Gemini CLI 0.16.0 was probed first: it asked for a browser login, so it was not run and is not counted. The protocol was declared after the controls and before the first session."
      ],
      "caveats": [
        "Codex CLI ran with the tester’s global AGENTS.md: --ignore-user-config does not switch that file off. 12 of 12 Codex sessions read the tester’s notes, 10 wrote a WORKLOG.md and 9 reported a commit attempt. Its time, tool calls and lines changed include that work. Claude Code ran with setting sources off, and no Claude session did any of it.",
        "Every agent passed every session, so the pass rates sit at the 100% ceiling. These tasks are too easy to separate the agents on quality; only time, tool use, tokens and diff size differ.",
        "Each run pairs a CLI with a model (Claude Code with Claude models, Codex CLI with GPT-6.1 Sol), so the results cannot separate the CLI from the model.",
        "n = 12 sessions per agent (2 per task). Time ranges are the fastest and slowest sessions, not confidence intervals; the two lanes shared one machine.",
        "Tokens are as each CLI reports them, with different tokenizers and context handling: compare tokens inside Claude Code (Sonnet vs Opus), not across vendors. Costs are list-price calculations on subscription sessions, not invoices."
      ],
      "sourceIds": [
        "agent-coding-agents",
        "calc-repricing",
        "price-anthropic",
        "price-openai"
      ],
      "stats": [
        {
          "id": "coding-agents-pass-all",
          "label": "Sessions that passed every hidden check, all three agents",
          "value": 1,
          "unit": "rate",
          "display": "100% (36/36)",
          "n": 36,
          "ci": [
            0.9036,
            1
          ],
          "note": "The ceiling: 36 of 36 means this task set cannot rank the agents on quality."
        },
        {
          "id": "coding-agents-sessions",
          "label": "Graded sessions (6 tasks × 2 repetitions × 3 agents)",
          "value": 36,
          "unit": "count",
          "display": "36",
          "n": 36,
          "note": "0 timeouts, 0 errors or usage-limit stops, nothing retried. Gemini CLI not run: it asked for a browser login."
        },
        {
          "id": "coding-agents-median-time-fastest",
          "label": "Median time per session, Sonnet 5.5 in Claude Code",
          "value": 23.1,
          "unit": "seconds",
          "display": "23.1 s",
          "n": 12,
          "note": "Fastest 18.7 s, slowest 44.5 s (a range of 12 sessions, not an interval)."
        },
        {
          "id": "coding-agents-time-ratio",
          "label": "Median time, GPT-6.1 Sol in Codex CLI vs Sonnet 5.5 in Claude Code (ratio of medians)",
          "value": 4.92,
          "unit": "ratio",
          "display": "4.9×",
          "n": 12,
          "note": "113.4 s vs 23.1 s. The run ranges do not overlap. Codex time includes the work its standing instructions asked for (see the caveats)."
        },
        {
          "id": "coding-agents-tester-notes",
          "label": "Codex sessions that read the tester’s global notes (standing instructions)",
          "value": 12,
          "unit": "count",
          "display": "12 of 12",
          "n": 12,
          "note": "10 of 12 also wrote a WORKLOG.md and 9 reported a commit attempt; no task asked for either. No Claude Code session did any of this."
        },
        {
          "id": "coding-agents-outside-edits",
          "label": "Edits that landed outside the task repository",
          "value": 0,
          "unit": "count",
          "display": "0",
          "n": 36,
          "note": "3 attempted writes to a temp folder (1 Sonnet 5.5, 2 Opus 5.5) were refused by the Claude Code permission check."
        },
        {
          "id": "coding-agents-cost-total",
          "label": "List-price estimate of all 36 sessions (calculation, not an invoice)",
          "value": 4.87,
          "unit": "usd",
          "display": "$4.87",
          "n": 36
        }
      ],
      "charts": [
        {
          "id": "coding-agents-pass-rate",
          "title": "Coding sessions that passed every hidden check",
          "subtitle": "A pass needs every hidden check · 12 sessions per agent · 95% Wilson intervals",
          "kind": "dot-range",
          "unit": "rate",
          "whisker": "ci95",
          "yLabel": "Passed",
          "viz": "IntervalDotPlot",
          "series": [
            {
              "name": "Passed every hidden check",
              "points": [
                {
                  "label": "Claude Sonnet 5.5 · Claude Code",
                  "value": 1,
                  "lo": 0.7575,
                  "hi": 1,
                  "n": 12
                },
                {
                  "label": "Claude Opus 5.5 · Claude Code",
                  "value": 1,
                  "lo": 0.7575,
                  "hi": 1,
                  "n": 12
                },
                {
                  "label": "GPT-6.1 Sol (medium, tester’s AGENTS.md) · Codex CLI",
                  "value": 1,
                  "lo": 0.7575,
                  "hi": 1,
                  "n": 12
                }
              ]
            }
          ],
          "note": "6 tasks × 2 repetitions per agent. All 36 sessions passed, so the chart sits at its ceiling: this task set cannot rank the agents on quality. Before the first session, every base repository failed its hidden checks and every reference patch passed them.",
          "sourceIds": [
            "agent-coding-agents"
          ]
        },
        {
          "id": "coding-agents-wall-time",
          "title": "Time per coding session",
          "subtitle": "Median wall time; whiskers = fastest and slowest of 12 sessions (not an interval)",
          "kind": "dot-range",
          "unit": "seconds",
          "whisker": "minmax",
          "yLabel": "Seconds",
          "viz": "LatencyLanes",
          "series": [
            {
              "name": "Wall time per session",
              "points": [
                {
                  "label": "Claude Sonnet 5.5 · Claude Code",
                  "value": 23.1,
                  "lo": 18.7,
                  "hi": 44.5,
                  "n": 12,
                  "highlight": true
                },
                {
                  "label": "Claude Opus 5.5 · Claude Code",
                  "value": 56.9,
                  "lo": 29.8,
                  "hi": 185.8,
                  "n": 12
                },
                {
                  "label": "GPT-6.1 Sol (medium, tester’s AGENTS.md) · Codex CLI",
                  "value": 113.4,
                  "lo": 78.5,
                  "hi": 221.9,
                  "n": 12
                }
              ]
            }
          ],
          "note": "CLI process start to exit. One session at a time per lane; the Claude Code and Codex CLI lanes ran at the same time on one machine with different accounts. Codex time includes reading the tester’s notes, writing a work log and trying to commit. A range is not a confidence interval.",
          "sourceIds": [
            "agent-coding-agents"
          ]
        },
        {
          "id": "coding-agents-time-by-task",
          "title": "Time per coding task",
          "subtitle": "Median of 2 sessions per task and agent, seconds",
          "kind": "grouped-bar",
          "unit": "seconds",
          "xLabel": "Task",
          "yLabel": "Seconds",
          "viz": "SmallMultiplesBars",
          "series": [
            {
              "name": "Claude Sonnet 5.5 · Claude Code",
              "points": [
                {
                  "label": "Pagination fix",
                  "value": 19.1,
                  "n": 2
                },
                {
                  "label": "CLI --top flag",
                  "value": 43.6,
                  "n": 2
                },
                {
                  "label": "Invoice refactor",
                  "value": 24.1,
                  "n": 2
                },
                {
                  "label": "LRU cache",
                  "value": 23.6,
                  "n": 2
                },
                {
                  "label": "Queue race",
                  "value": 22.7,
                  "n": 2
                },
                {
                  "label": "Strict TypeScript types",
                  "value": 27.6,
                  "n": 2
                }
              ]
            },
            {
              "name": "Claude Opus 5.5 · Claude Code",
              "points": [
                {
                  "label": "Pagination fix",
                  "value": 39.3,
                  "n": 2
                },
                {
                  "label": "CLI --top flag",
                  "value": 55.3,
                  "n": 2
                },
                {
                  "label": "Invoice refactor",
                  "value": 58.4,
                  "n": 2
                },
                {
                  "label": "LRU cache",
                  "value": 62.4,
                  "n": 2
                },
                {
                  "label": "Queue race",
                  "value": 118.6,
                  "n": 2
                },
                {
                  "label": "Strict TypeScript types",
                  "value": 58,
                  "n": 2
                }
              ]
            },
            {
              "name": "GPT-6.1 Sol (medium, tester’s AGENTS.md) · Codex CLI",
              "points": [
                {
                  "label": "Pagination fix",
                  "value": 82.7,
                  "n": 2
                },
                {
                  "label": "CLI --top flag",
                  "value": 101.2,
                  "n": 2
                },
                {
                  "label": "Invoice refactor",
                  "value": 129.1,
                  "n": 2
                },
                {
                  "label": "LRU cache",
                  "value": 203.1,
                  "n": 2
                },
                {
                  "label": "Queue race",
                  "value": 98.6,
                  "n": 2
                },
                {
                  "label": "Strict TypeScript types",
                  "value": 143.6,
                  "n": 2
                }
              ]
            }
          ],
          "note": "2 sessions per cell, so one slow session moves a bar: Opus 5.5 in Claude Code took 51.5 s and 185.8 s on Queue race.",
          "sourceIds": [
            "agent-coding-agents"
          ]
        },
        {
          "id": "coding-agents-tool-calls",
          "title": "Tool calls per coding session",
          "subtitle": "Median; whiskers = fewest and most of 12 sessions (not an interval)",
          "kind": "dot-range",
          "unit": "calls",
          "whisker": "minmax",
          "yLabel": "Tool calls",
          "viz": "IntervalDotPlot",
          "series": [
            {
              "name": "Tool calls per session",
              "points": [
                {
                  "label": "Claude Sonnet 5.5 · Claude Code",
                  "value": 7.5,
                  "lo": 3,
                  "hi": 14,
                  "n": 12
                },
                {
                  "label": "Claude Opus 5.5 · Claude Code",
                  "value": 7.5,
                  "lo": 5,
                  "hi": 14,
                  "n": 12
                },
                {
                  "label": "GPT-6.1 Sol (medium, tester’s AGENTS.md) · Codex CLI",
                  "value": 12.5,
                  "lo": 8,
                  "hi": 18,
                  "n": 12
                }
              ]
            }
          ],
          "note": "Claude Code counts its tool calls (Bash, Read, Edit, Write, Glob, Grep). Codex CLI counts shell commands and file changes; it has no separate read tool, so it reads files with shell commands. Turns are not compared: Codex reports one turn per run.",
          "sourceIds": [
            "agent-coding-agents"
          ]
        },
        {
          "id": "coding-agents-token-mix",
          "title": "Tokens per session, as each CLI reports them",
          "subtitle": "Mean per session by kind · not comparable across vendors",
          "kind": "stacked-bar",
          "unit": "tokens",
          "yLabel": "Tokens",
          "viz": "TokenStack",
          "series": [
            {
              "name": "Cache read",
              "points": [
                {
                  "label": "Claude Sonnet 5.5 · Claude Code",
                  "value": 73846,
                  "n": 12
                },
                {
                  "label": "Claude Opus 5.5 · Claude Code",
                  "value": 96964,
                  "n": 12
                },
                {
                  "label": "GPT-6.1 Sol (medium, tester’s AGENTS.md) · Codex CLI",
                  "value": 174560,
                  "n": 12
                }
              ]
            },
            {
              "name": "Cache write",
              "points": [
                {
                  "label": "Claude Sonnet 5.5 · Claude Code",
                  "value": 9163,
                  "n": 12
                },
                {
                  "label": "Claude Opus 5.5 · Claude Code",
                  "value": 11378,
                  "n": 12
                }
              ]
            },
            {
              "name": "Uncached input",
              "points": [
                {
                  "label": "Claude Sonnet 5.5 · Claude Code",
                  "value": 13,
                  "n": 12
                },
                {
                  "label": "Claude Opus 5.5 · Claude Code",
                  "value": 16,
                  "n": 12
                },
                {
                  "label": "GPT-6.1 Sol (medium, tester’s AGENTS.md) · Codex CLI",
                  "value": 19811,
                  "n": 12
                }
              ]
            },
            {
              "name": "Output",
              "points": [
                {
                  "label": "Claude Sonnet 5.5 · Claude Code",
                  "value": 3359,
                  "n": 12
                },
                {
                  "label": "Claude Opus 5.5 · Claude Code",
                  "value": 5621,
                  "n": 12
                },
                {
                  "label": "GPT-6.1 Sol (medium, tester’s AGENTS.md) · Codex CLI",
                  "value": 4072,
                  "n": 12
                }
              ]
            }
          ],
          "note": "Claude Code reports uncached input, cache reads, cache writes and output. Codex CLI reports input (its cached part included), the cached part and output (reasoning included), and no cache writes. Different tokenizers and context handling: compare Sonnet with Opus here, not Claude with Codex.",
          "sourceIds": [
            "agent-coding-agents"
          ]
        },
        {
          "id": "coding-agents-diff-lines",
          "title": "Lines changed per session, by kind of file",
          "subtitle": "Mean lines added plus deleted per session, against the base commit",
          "kind": "stacked-bar",
          "unit": "count",
          "yLabel": "Lines",
          "viz": "TokenStack",
          "series": [
            {
              "name": "Code the task is about",
              "points": [
                {
                  "label": "Claude Sonnet 5.5 · Claude Code",
                  "value": 48.4,
                  "n": 12
                },
                {
                  "label": "Claude Opus 5.5 · Claude Code",
                  "value": 56.8,
                  "n": 12
                },
                {
                  "label": "GPT-6.1 Sol (medium, tester’s AGENTS.md) · Codex CLI",
                  "value": 52.7,
                  "n": 12
                }
              ]
            },
            {
              "name": "Tests",
              "points": [
                {
                  "label": "Claude Sonnet 5.5 · Claude Code",
                  "value": 2.4,
                  "n": 12
                },
                {
                  "label": "Claude Opus 5.5 · Claude Code",
                  "value": 19.4,
                  "n": 12
                },
                {
                  "label": "GPT-6.1 Sol (medium, tester’s AGENTS.md) · Codex CLI",
                  "value": 70.4,
                  "n": 12
                }
              ]
            },
            {
              "name": "README and package.json",
              "points": [
                {
                  "label": "Claude Sonnet 5.5 · Claude Code",
                  "value": 0.8,
                  "n": 12
                },
                {
                  "label": "Claude Opus 5.5 · Claude Code",
                  "value": 3.3,
                  "n": 12
                },
                {
                  "label": "GPT-6.1 Sol (medium, tester’s AGENTS.md) · Codex CLI",
                  "value": 4.6,
                  "n": 12
                }
              ]
            },
            {
              "name": "Work log and notes nobody asked for",
              "points": [
                {
                  "label": "Claude Sonnet 5.5 · Claude Code",
                  "value": 0,
                  "n": 12
                },
                {
                  "label": "Claude Opus 5.5 · Claude Code",
                  "value": 0,
                  "n": 12
                },
                {
                  "label": "GPT-6.1 Sol (medium, tester’s AGENTS.md) · Codex CLI",
                  "value": 9.4,
                  "n": 12
                }
              ]
            }
          ],
          "note": "Counted from each session’s patch, new files included. Median lines per session: Sonnet 5.5 in Claude Code 45, Opus 5.5 in Claude Code 77.5, GPT-6.1 Sol in Codex CLI 94. Codex wrote a WORKLOG.md in 10 of 12 sessions; no task asked for one.",
          "sourceIds": [
            "agent-coding-agents"
          ]
        },
        {
          "id": "coding-agents-cost-per-pass",
          "title": "List-price cost per passing coding session (calculation)",
          "subtitle": "Reported tokens of all 12 sessions × list price, divided by the passes",
          "kind": "bar",
          "unit": "usd",
          "yLabel": "USD per pass",
          "viz": "CostBars",
          "series": [
            {
              "name": "List-price cost per pass",
              "points": [
                {
                  "label": "Claude Sonnet 5.5 · Claude Code",
                  "value": 0.085,
                  "n": 12
                },
                {
                  "label": "Claude Opus 5.5 · Claude Code",
                  "value": 0.2229,
                  "n": 12
                },
                {
                  "label": "GPT-6.1 Sol (medium, tester’s AGENTS.md) · Codex CLI",
                  "value": 0.0978,
                  "n": 12
                }
              ]
            }
          ],
          "note": "Calculation, not a bill: both CLIs ran on flat subscriptions. Claude cache writes are priced at 2× input, as in the other studies (Claude Code’s own estimate gives the same totals); Codex cached input at its cache-read price. Codex input includes its own system prompt and, here, the tester’s AGENTS.md.",
          "sourceIds": [
            "agent-coding-agents",
            "calc-repricing",
            "price-anthropic",
            "price-openai"
          ]
        }
      ],
      "tables": [
        {
          "id": "coding-agents-cells",
          "title": "Every task and agent: sessions that passed (of 2)",
          "viz": "HeatMatrix",
          "columns": [
            {
              "key": "task",
              "label": "Task",
              "unit": "text"
            },
            {
              "key": "a0",
              "label": "Sonnet 5.5 in Claude Code",
              "unit": "text"
            },
            {
              "key": "a1",
              "label": "Opus 5.5 in Claude Code",
              "unit": "text"
            },
            {
              "key": "a2",
              "label": "GPT-6.1 Sol in Codex CLI",
              "unit": "text"
            },
            {
              "key": "checks",
              "label": "Hidden checks",
              "unit": "text"
            }
          ],
          "rows": [
            {
              "task": "Pagination fix",
              "a0": "2/2",
              "a1": "2/2",
              "a2": "2/2",
              "checks": "13 hidden tests"
            },
            {
              "task": "CLI --top flag",
              "a0": "2/2",
              "a1": "2/2",
              "a2": "2/2",
              "checks": "11 hidden tests that run the CLI"
            },
            {
              "task": "Invoice refactor",
              "a0": "2/2",
              "a1": "2/2",
              "a2": "2/2",
              "checks": "20 golden behaviour tests and 4 structure tests"
            },
            {
              "task": "LRU cache",
              "a0": "2/2",
              "a1": "2/2",
              "a2": "2/2",
              "checks": "16 hidden tests"
            },
            {
              "task": "Queue race",
              "a0": "2/2",
              "a1": "2/2",
              "a2": "2/2",
              "checks": "10 hidden tests"
            },
            {
              "task": "Strict TypeScript types",
              "a0": "2/2",
              "a1": "2/2",
              "a2": "2/2",
              "checks": "static checks, tsc with a hidden usage file (11 @ts-expect-error probes) and 4 runtime tests"
            }
          ]
        },
        {
          "id": "coding-agents-tasks",
          "title": "The six tasks and their controls",
          "columns": [
            {
              "key": "task",
              "label": "Task",
              "unit": "text"
            },
            {
              "key": "what",
              "label": "What the agent had to do",
              "unit": "text"
            },
            {
              "key": "checks",
              "label": "Hidden checks",
              "unit": "text"
            },
            {
              "key": "base",
              "label": "Base repository",
              "unit": "text"
            },
            {
              "key": "reference",
              "label": "Reference patch",
              "unit": "text"
            }
          ],
          "rows": [
            {
              "task": "Pagination fix",
              "what": "Fix a pagination bug that spans two files (offset, page count, next and previous flags, validation) to the README rules",
              "checks": "13 hidden tests",
              "base": "fails (tests 1/13)",
              "reference": "passes (tests 13/13)"
            },
            {
              "task": "CLI --top flag",
              "what": "Add a --top <n> flag with strict validation and exact error text to a small CLI",
              "checks": "11 hidden tests that run the CLI",
              "base": "fails (tests 2/11)",
              "reference": "passes (tests 11/11)"
            },
            {
              "task": "Invoice refactor",
              "what": "Remove duplication without a change in behaviour, floating-point and validation-order quirks included",
              "checks": "20 golden behaviour tests and 4 structure tests",
              "base": "fails (tests 21/24)",
              "reference": "passes (tests 24/24)"
            },
            {
              "task": "LRU cache",
              "what": "Implement an LRU cache to a written spec (recency, eviction callback, TTL with an injected clock, SameValueZero keys, O(1) get and set)",
              "checks": "16 hidden tests",
              "base": "fails (tests 0/16)",
              "reference": "passes (tests 16/16)"
            },
            {
              "task": "Queue race",
              "what": "Fix a concurrency race and an onIdle hang in an async task queue",
              "checks": "10 hidden tests",
              "base": "fails (tests 4/10)",
              "reference": "passes (tests 10/10)"
            },
            {
              "task": "Strict TypeScript types",
              "what": "Make tsc pass on strict settings with no any and no suppressions; the usage file and the config stay unchanged",
              "checks": "static checks, tsc with a hidden usage file (11 @ts-expect-error probes) and 4 runtime tests",
              "base": "fails (tests 4/4; static checks fail)",
              "reference": "passes (tests 4/4)"
            }
          ]
        }
      ],
      "related": [
        "hard-model-head-to-head",
        "model-head-to-head",
        "swe-bench-opus-vs-sonnet",
        "effort-ladder"
      ],
      "hero": {
        "statIds": [
          "coding-agents-time-ratio"
        ]
      }
    },
    {
      "slug": "effort-ladder",
      "title": "Does more effort buy quality? Sonnet, Opus and GPT-6.1 Sol on 8 hard tasks",
      "seoTitle": "Effort ladder: does more AI effort buy quality?",
      "description": "176 calls on 8 hard tasks at low, medium, high and default effort: Sonnet, Opus and GPT-6.1 Sol. Pass rate with 95% intervals, time, tokens, cost per pass.",
      "question": "On 8 hard tasks with strict validators, does a higher effort setting buy a higher pass rate for Sonnet, Opus and GPT-6.1 Sol, and what does it cost in time, tokens and list price per pass?",
      "answer": "Not on this set. Every one of the 11 configurations (Sonnet at low, medium, high and default, Opus at low, medium, high and default and GPT-6.1 Sol through Codex CLI at low, medium and high) passed all 16 calls strictly (95% interval 81% to 100% each), so pass rate does not separate any effort level. The hard set has a ceiling for these models: with 16 calls per cell it cannot rule out a difference of up to about 19 points. What more effort did change is output and time. Median total time per call by effort: Sonnet: low 5.8 s, medium 7.6 s, high 8.8 s, default 8.0 s (median output tokens 667 / 770 / 1,192 / 1,054); Opus: low 7.5 s, medium 9.7 s, high 10.1 s, default 9.2 s (median output tokens 594 / 853 / 1,052 / 945); GPT-6.1 Sol (Codex CLI): low 13.6 s, medium 13.1 s, high 18.1 s (median output tokens 284 / 335 / 436). Within each model, the fastest and slowest calls of every effort overlap, so these medians describe this run; they are not a tested ranking. At list price (a calculation; the calls ran on subscriptions), cost per strict pass went from $0.0122 at low to $0.0167 at high for Sonnet, $0.0212 at low to $0.0337 at high for Opus and $0.0128 at low to $0.0151 at high for GPT-6.1 Sol (Codex CLI).",
      "date": "2026-10-06",
      "updated": "2026-10-06",
      "tags": [
        "effort",
        "reasoning-effort",
        "hard-tasks",
        "claude-sonnet",
        "claude-opus",
        "gpt-6-1-sol",
        "claude-code",
        "codex-cli",
        "latency",
        "tokens"
      ],
      "method": [
        "Protocols declared before the first call, one per route. A follow-up to the hard head-to-head (/benchmarks/hard-model-head-to-head), on the same 8 tasks, validators, controls and CLI flags.",
        "New cells: Claude Sonnet 5.5 (low) · Claude Code; Claude Sonnet 5.5 (medium) · Claude Code; Claude Sonnet 5.5 (high) · Claude Code; Claude Opus 5.5 (low) · Claude Code; Claude Opus 5.5 (medium) · Claude Code; GPT-6.1 Sol (low) · Codex CLI. 96 calls (8 tasks × 2 repetitions per cell), one call at a time per account; order rep-major, then task, then configuration.",
        "Reference cells (5): Claude Sonnet 5.5 · Claude Code; Claude Opus 5.5 (high) · Claude Code; Claude Opus 5.5 · Claude Code; GPT-6.1 Sol (medium) · Codex CLI; GPT-6.1 Sol (high) · Codex CLI, from the hard head-to-head. Reused, not rerun. Claude reference cells keep repetitions 1-2 (n = 16) so every cell has the same design; all 24 of their calls passed in the hard study.",
        "Effort: \"--effort <level>\" for Claude Code and the thread effort for Codex CLI. \"default\" means the flag was not passed and the CLI chose the level.",
        "Controls before inference: 8/8 reference answers pass, 26/26 plausible wrong answers fail, and 8/8 wrapped references are flagged as format misses.",
        "Strict pass, format miss and wrong answer as in the hard head-to-head. Isolation: fresh empty working folder, tools off, no MCP servers, no session persistence, one turn, 300 s timeout per call.",
        "Stop rules: stop at the first usage-limit or rate-limit text. No batch stopped early, nothing was trimmed or retried, and no call failed.",
        "Cost per strict pass: list price × reported tokens for every call in the cell, divided by its strict passes. A calculation."
      ],
      "caveats": [
        "Every cell passed every call, so the set has a ceiling: a perfect 16/16 has a 95% interval of 81% to 100%. This study cannot show that effort does not matter on harder work; it shows that these 8 tasks do not need more than low effort.",
        "Reference cells ran in a different batch and hour than the new cells (provider load can differ), on the same host, CLI versions, cases and flags.",
        "Each cell has only 2 calls per task. Medians of 16 calls move with a few slow calls; the ranges are wide and overlap.",
        "Claude Code and Codex CLI rows pair a CLI with a model; each CLI adds its own system prompt and start-up time. Effort levels are not the same scale across vendors.",
        "List-price costs are calculations; the calls used flat subscriptions."
      ],
      "sourceIds": [
        "agent-effort-ladder",
        "calc-repricing",
        "price-anthropic",
        "price-openai"
      ],
      "stats": [
        {
          "id": "effort-ladder-pass-new",
          "label": "New effort-ladder calls that passed strictly",
          "value": 1,
          "unit": "rate",
          "display": "100% (96/96)",
          "n": 96,
          "ci": [
            0.9615,
            1
          ]
        },
        {
          "id": "effort-ladder-pass-all",
          "label": "All effort-ladder calls that passed strictly, reference cells included",
          "value": 1,
          "unit": "rate",
          "display": "100% (176/176)",
          "n": 176,
          "ci": [
            0.9786,
            1
          ]
        },
        {
          "id": "effort-ladder-cells",
          "label": "Configurations on the ladder (new + reference)",
          "value": 11,
          "unit": "count",
          "display": "11 (6 new, 5 reference)",
          "n": 176
        },
        {
          "id": "effort-ladder-format-misses",
          "label": "Format misses and wrong answers on the ladder",
          "value": 0,
          "unit": "count",
          "display": "0 format misses, 0 wrong answers",
          "n": 176
        }
      ],
      "charts": [
        {
          "id": "effort-ladder-pass-rate",
          "title": "Strict pass rate by effort on eight hard tasks",
          "subtitle": "Every configuration: 8 tasks × 2 repetitions. Whiskers are 95% Wilson intervals",
          "kind": "dot-range",
          "unit": "rate",
          "yLabel": "Passed",
          "series": [
            {
              "name": "Strict pass",
              "points": [
                {
                  "label": "Claude Sonnet 5.5 (low) · Claude Code",
                  "value": 1,
                  "lo": 0.8064,
                  "hi": 1,
                  "n": 16
                },
                {
                  "label": "Claude Sonnet 5.5 (medium) · Claude Code",
                  "value": 1,
                  "lo": 0.8064,
                  "hi": 1,
                  "n": 16
                },
                {
                  "label": "Claude Sonnet 5.5 (high) · Claude Code",
                  "value": 1,
                  "lo": 0.8064,
                  "hi": 1,
                  "n": 16
                },
                {
                  "label": "Claude Sonnet 5.5 · Claude Code",
                  "value": 1,
                  "lo": 0.8064,
                  "hi": 1,
                  "n": 16
                },
                {
                  "label": "Claude Opus 5.5 (low) · Claude Code",
                  "value": 1,
                  "lo": 0.8064,
                  "hi": 1,
                  "n": 16
                },
                {
                  "label": "Claude Opus 5.5 (medium) · Claude Code",
                  "value": 1,
                  "lo": 0.8064,
                  "hi": 1,
                  "n": 16
                },
                {
                  "label": "Claude Opus 5.5 (high) · Claude Code",
                  "value": 1,
                  "lo": 0.8064,
                  "hi": 1,
                  "n": 16
                },
                {
                  "label": "Claude Opus 5.5 · Claude Code",
                  "value": 1,
                  "lo": 0.8064,
                  "hi": 1,
                  "n": 16
                },
                {
                  "label": "GPT-6.1 Sol (low) · Codex CLI",
                  "value": 1,
                  "lo": 0.8064,
                  "hi": 1,
                  "n": 16
                },
                {
                  "label": "GPT-6.1 Sol (medium) · Codex CLI",
                  "value": 1,
                  "lo": 0.8064,
                  "hi": 1,
                  "n": 16
                },
                {
                  "label": "GPT-6.1 Sol (high) · Codex CLI",
                  "value": 1,
                  "lo": 0.8064,
                  "hi": 1,
                  "n": 16
                }
              ]
            }
          ],
          "note": "Whiskers are 95% Wilson intervals. A format miss never counts as a pass. Reference cells (efforts the hard head-to-head already ran) are reused, not rerun; Claude reference cells keep repetitions 1-2, so every cell is 8 tasks × 2 repetitions.",
          "whisker": "ci95",
          "sourceIds": [
            "agent-effort-ladder"
          ]
        },
        {
          "id": "effort-ladder-total-latency",
          "title": "Total time per call by effort on hard tasks",
          "subtitle": "Median per configuration; whiskers = fastest and slowest call",
          "kind": "dot-range",
          "unit": "seconds",
          "yLabel": "Seconds",
          "series": [
            {
              "name": "Total time per call by effort on hard tasks",
              "points": [
                {
                  "label": "Claude Sonnet 5.5 (low) · Claude Code",
                  "value": 5.82,
                  "lo": 2.78,
                  "hi": 19.96,
                  "n": 16
                },
                {
                  "label": "Claude Sonnet 5.5 (medium) · Claude Code",
                  "value": 7.63,
                  "lo": 2.71,
                  "hi": 24.01,
                  "n": 16
                },
                {
                  "label": "Claude Sonnet 5.5 (high) · Claude Code",
                  "value": 8.81,
                  "lo": 2.93,
                  "hi": 35.81,
                  "n": 16
                },
                {
                  "label": "Claude Sonnet 5.5 · Claude Code",
                  "value": 7.97,
                  "lo": 2.26,
                  "hi": 21.61,
                  "n": 16
                },
                {
                  "label": "Claude Opus 5.5 (low) · Claude Code",
                  "value": 7.5,
                  "lo": 3.34,
                  "hi": 15.82,
                  "n": 16
                },
                {
                  "label": "Claude Opus 5.5 (medium) · Claude Code",
                  "value": 9.72,
                  "lo": 4.78,
                  "hi": 31.36,
                  "n": 16
                },
                {
                  "label": "Claude Opus 5.5 (high) · Claude Code",
                  "value": 10.11,
                  "lo": 3.63,
                  "hi": 63,
                  "n": 16
                },
                {
                  "label": "Claude Opus 5.5 · Claude Code",
                  "value": 9.18,
                  "lo": 4.24,
                  "hi": 27.21,
                  "n": 16
                },
                {
                  "label": "GPT-6.1 Sol (low) · Codex CLI",
                  "value": 13.62,
                  "lo": 7.94,
                  "hi": 44.29,
                  "n": 16
                },
                {
                  "label": "GPT-6.1 Sol (medium) · Codex CLI",
                  "value": 13.11,
                  "lo": 8.54,
                  "hi": 61.6,
                  "n": 16
                },
                {
                  "label": "GPT-6.1 Sol (high) · Codex CLI",
                  "value": 18.12,
                  "lo": 11.67,
                  "hi": 92.21,
                  "n": 16
                }
              ]
            }
          ],
          "note": "Whiskers are a range (fastest and slowest call), not a confidence interval. One host, one network. Reference cells (efforts the hard head-to-head already ran) are reused, not rerun; Claude reference cells keep repetitions 1-2, so every cell is 8 tasks × 2 repetitions. The reference cells ran in a different hour.",
          "whisker": "minmax",
          "sourceIds": [
            "agent-effort-ladder"
          ]
        },
        {
          "id": "effort-ladder-time-by-effort",
          "title": "Median total time per call, by effort",
          "subtitle": "One line per model and route; default = the effort flag was not passed",
          "kind": "line",
          "unit": "seconds",
          "xLabel": "Effort",
          "yLabel": "Seconds (median)",
          "series": [
            {
              "name": "Claude Sonnet 5.5 · Claude Code",
              "points": [
                {
                  "label": "low",
                  "value": 5.82,
                  "n": 16
                },
                {
                  "label": "medium",
                  "value": 7.63,
                  "n": 16
                },
                {
                  "label": "high",
                  "value": 8.81,
                  "n": 16
                },
                {
                  "label": "default",
                  "value": 7.97,
                  "n": 16
                }
              ]
            },
            {
              "name": "Claude Opus 5.5 · Claude Code",
              "points": [
                {
                  "label": "low",
                  "value": 7.5,
                  "n": 16
                },
                {
                  "label": "medium",
                  "value": 9.72,
                  "n": 16
                },
                {
                  "label": "high",
                  "value": 10.11,
                  "n": 16
                },
                {
                  "label": "default",
                  "value": 9.18,
                  "n": 16
                }
              ]
            },
            {
              "name": "GPT-6.1 Sol · Codex CLI",
              "points": [
                {
                  "label": "low",
                  "value": 13.62,
                  "n": 16
                },
                {
                  "label": "medium",
                  "value": 13.11,
                  "n": 16
                },
                {
                  "label": "high",
                  "value": 18.12,
                  "n": 16
                }
              ]
            }
          ],
          "note": "Medians only; the per-call ranges are in the total-time chart and they overlap. \"default\" is placed last because its level is not known: the CLI chose it.",
          "sourceIds": [
            "agent-effort-ladder"
          ]
        },
        {
          "id": "effort-ladder-output-tokens",
          "title": "Output tokens per call by effort on hard tasks",
          "subtitle": "Median per configuration; reasoning tokens as the CLI reports them",
          "kind": "grouped-bar",
          "unit": "tokens",
          "yLabel": "Tokens",
          "series": [
            {
              "name": "Output tokens",
              "points": [
                {
                  "label": "Claude Sonnet 5.5 (low) · Claude Code",
                  "value": 667,
                  "n": 16
                },
                {
                  "label": "Claude Sonnet 5.5 (medium) · Claude Code",
                  "value": 770,
                  "n": 16
                },
                {
                  "label": "Claude Sonnet 5.5 (high) · Claude Code",
                  "value": 1192,
                  "n": 16
                },
                {
                  "label": "Claude Sonnet 5.5 · Claude Code",
                  "value": 1054,
                  "n": 16
                },
                {
                  "label": "Claude Opus 5.5 (low) · Claude Code",
                  "value": 594,
                  "n": 16
                },
                {
                  "label": "Claude Opus 5.5 (medium) · Claude Code",
                  "value": 853,
                  "n": 16
                },
                {
                  "label": "Claude Opus 5.5 (high) · Claude Code",
                  "value": 1052,
                  "n": 16
                },
                {
                  "label": "Claude Opus 5.5 · Claude Code",
                  "value": 945,
                  "n": 16
                },
                {
                  "label": "GPT-6.1 Sol (low) · Codex CLI",
                  "value": 284,
                  "n": 16
                },
                {
                  "label": "GPT-6.1 Sol (medium) · Codex CLI",
                  "value": 335,
                  "n": 16
                },
                {
                  "label": "GPT-6.1 Sol (high) · Codex CLI",
                  "value": 436,
                  "n": 16
                }
              ]
            },
            {
              "name": "Reasoning tokens",
              "points": [
                {
                  "label": "Claude Sonnet 5.5 (low) · Claude Code",
                  "value": 273,
                  "n": 16
                },
                {
                  "label": "Claude Sonnet 5.5 (medium) · Claude Code",
                  "value": 422,
                  "n": 16
                },
                {
                  "label": "Claude Sonnet 5.5 (high) · Claude Code",
                  "value": 745,
                  "n": 16
                },
                {
                  "label": "Claude Sonnet 5.5 · Claude Code",
                  "value": 668,
                  "n": 16
                },
                {
                  "label": "Claude Opus 5.5 (low) · Claude Code",
                  "value": 87,
                  "n": 16
                },
                {
                  "label": "Claude Opus 5.5 (medium) · Claude Code",
                  "value": 518,
                  "n": 16
                },
                {
                  "label": "Claude Opus 5.5 (high) · Claude Code",
                  "value": 614,
                  "n": 16
                },
                {
                  "label": "Claude Opus 5.5 · Claude Code",
                  "value": 538,
                  "n": 16
                },
                {
                  "label": "GPT-6.1 Sol (low) · Codex CLI",
                  "value": 63,
                  "n": 16
                },
                {
                  "label": "GPT-6.1 Sol (medium) · Codex CLI",
                  "value": 150,
                  "n": 16
                },
                {
                  "label": "GPT-6.1 Sol (high) · Codex CLI",
                  "value": 225,
                  "n": 16
                }
              ]
            }
          ],
          "note": "Reasoning tokens are part of the output tokens where the CLI reports them; 0 can mean \"not reported\". Their content is never captured. More tokens is not better or worse by itself.",
          "sourceIds": [
            "agent-effort-ladder"
          ]
        },
        {
          "id": "effort-ladder-cost-per-pass",
          "title": "List-price cost per strict pass by effort (calculation)",
          "subtitle": "All calls in a configuration divided by its strict passes",
          "kind": "bar",
          "unit": "usd",
          "yLabel": "USD per strict pass",
          "series": [
            {
              "name": "Cost per strict pass",
              "points": [
                {
                  "label": "Claude Sonnet 5.5 (low) · Claude Code",
                  "value": 0.01219,
                  "n": 16
                },
                {
                  "label": "Claude Sonnet 5.5 (medium) · Claude Code",
                  "value": 0.01352,
                  "n": 16
                },
                {
                  "label": "Claude Sonnet 5.5 (high) · Claude Code",
                  "value": 0.01671,
                  "n": 16
                },
                {
                  "label": "Claude Sonnet 5.5 · Claude Code",
                  "value": 0.01398,
                  "n": 16
                },
                {
                  "label": "Claude Opus 5.5 (low) · Claude Code",
                  "value": 0.02115,
                  "n": 16
                },
                {
                  "label": "Claude Opus 5.5 (medium) · Claude Code",
                  "value": 0.02947,
                  "n": 16
                },
                {
                  "label": "Claude Opus 5.5 (high) · Claude Code",
                  "value": 0.03368,
                  "n": 16
                },
                {
                  "label": "Claude Opus 5.5 · Claude Code",
                  "value": 0.02893,
                  "n": 16
                },
                {
                  "label": "GPT-6.1 Sol (low) · Codex CLI",
                  "value": 0.01284,
                  "n": 16
                },
                {
                  "label": "GPT-6.1 Sol (medium) · Codex CLI",
                  "value": 0.02564,
                  "n": 16
                },
                {
                  "label": "GPT-6.1 Sol (high) · Codex CLI",
                  "value": 0.01514,
                  "n": 16
                }
              ]
            }
          ],
          "note": "Calculation, not a bill: reported tokens × list price, cache reads and writes priced as in the hard head-to-head; the calls ran on flat subscriptions. Codex input includes its own system prompt and tool schemas, so a Claude-vs-Codex gap is partly the CLI.",
          "sourceIds": [
            "agent-effort-ladder",
            "calc-repricing",
            "price-anthropic",
            "price-openai"
          ]
        }
      ],
      "tables": [
        {
          "id": "effort-ladder-cells",
          "title": "Every effort-ladder cell",
          "columns": [
            {
              "key": "config",
              "label": "Configuration",
              "unit": "text"
            },
            {
              "key": "effort",
              "label": "Effort",
              "unit": "text"
            },
            {
              "key": "origin",
              "label": "Cell",
              "unit": "text"
            },
            {
              "key": "strict",
              "label": "Strict passes",
              "unit": "text"
            },
            {
              "key": "ci",
              "label": "95% interval",
              "unit": "text"
            },
            {
              "key": "formatMisses",
              "label": "Format misses",
              "unit": "count"
            },
            {
              "key": "wrong",
              "label": "Wrong answers",
              "unit": "count"
            },
            {
              "key": "medianTotal",
              "label": "Median total (s)",
              "unit": "seconds"
            },
            {
              "key": "rangeTotal",
              "label": "Fastest to slowest (s)",
              "unit": "text"
            },
            {
              "key": "medianOut",
              "label": "Median output tokens",
              "unit": "tokens"
            },
            {
              "key": "meanOut",
              "label": "Mean output tokens",
              "unit": "tokens"
            },
            {
              "key": "medianReasoning",
              "label": "Median reasoning tokens",
              "unit": "tokens"
            },
            {
              "key": "perPass",
              "label": "USD per strict pass (calculation)",
              "unit": "usd"
            }
          ],
          "rows": [
            {
              "config": "Claude Sonnet 5.5 (low) · Claude Code",
              "effort": "low",
              "origin": "new run",
              "strict": "16/16",
              "ci": "81% to 100%",
              "formatMisses": 0,
              "wrong": 0,
              "medianTotal": 5.82,
              "rangeTotal": "2.8 to 20",
              "medianOut": 667,
              "meanOut": 810,
              "medianReasoning": 273,
              "perPass": 0.01219
            },
            {
              "config": "Claude Sonnet 5.5 (medium) · Claude Code",
              "effort": "medium",
              "origin": "new run",
              "strict": "16/16",
              "ci": "81% to 100%",
              "formatMisses": 0,
              "wrong": 0,
              "medianTotal": 7.63,
              "rangeTotal": "2.7 to 24",
              "medianOut": 770,
              "meanOut": 966,
              "medianReasoning": 422,
              "perPass": 0.01352
            },
            {
              "config": "Claude Sonnet 5.5 (high) · Claude Code",
              "effort": "high",
              "origin": "new run",
              "strict": "16/16",
              "ci": "81% to 100%",
              "formatMisses": 0,
              "wrong": 0,
              "medianTotal": 8.81,
              "rangeTotal": "2.9 to 35.8",
              "medianOut": 1192,
              "meanOut": 1284,
              "medianReasoning": 745,
              "perPass": 0.01671
            },
            {
              "config": "Claude Sonnet 5.5 · Claude Code",
              "effort": "default",
              "origin": "reference (hard head-to-head)",
              "strict": "16/16",
              "ci": "81% to 100%",
              "formatMisses": 0,
              "wrong": 0,
              "medianTotal": 7.97,
              "rangeTotal": "2.3 to 21.6",
              "medianOut": 1054,
              "meanOut": 989,
              "medianReasoning": 668,
              "perPass": 0.01398
            },
            {
              "config": "Claude Opus 5.5 (low) · Claude Code",
              "effort": "low",
              "origin": "new run",
              "strict": "16/16",
              "ci": "81% to 100%",
              "formatMisses": 0,
              "wrong": 0,
              "medianTotal": 7.5,
              "rangeTotal": "3.3 to 15.8",
              "medianOut": 594,
              "meanOut": 665,
              "medianReasoning": 87,
              "perPass": 0.02115
            },
            {
              "config": "Claude Opus 5.5 (medium) · Claude Code",
              "effort": "medium",
              "origin": "new run",
              "strict": "16/16",
              "ci": "81% to 100%",
              "formatMisses": 0,
              "wrong": 0,
              "medianTotal": 9.72,
              "rangeTotal": "4.8 to 31.4",
              "medianOut": 853,
              "meanOut": 1104,
              "medianReasoning": 518,
              "perPass": 0.02947
            },
            {
              "config": "Claude Opus 5.5 (high) · Claude Code",
              "effort": "high",
              "origin": "reference (hard head-to-head)",
              "strict": "16/16",
              "ci": "81% to 100%",
              "formatMisses": 0,
              "wrong": 0,
              "medianTotal": 10.11,
              "rangeTotal": "3.6 to 63",
              "medianOut": 1052,
              "meanOut": 1314,
              "medianReasoning": 614,
              "perPass": 0.03368
            },
            {
              "config": "Claude Opus 5.5 · Claude Code",
              "effort": "default",
              "origin": "reference (hard head-to-head)",
              "strict": "16/16",
              "ci": "81% to 100%",
              "formatMisses": 0,
              "wrong": 0,
              "medianTotal": 9.18,
              "rangeTotal": "4.2 to 27.2",
              "medianOut": 945,
              "meanOut": 1053,
              "medianReasoning": 538,
              "perPass": 0.02893
            },
            {
              "config": "GPT-6.1 Sol (low) · Codex CLI",
              "effort": "low",
              "origin": "new run",
              "strict": "16/16",
              "ci": "81% to 100%",
              "formatMisses": 0,
              "wrong": 0,
              "medianTotal": 13.62,
              "rangeTotal": "7.9 to 44.3",
              "medianOut": 284,
              "meanOut": 420,
              "medianReasoning": 63,
              "perPass": 0.01284
            },
            {
              "config": "GPT-6.1 Sol (medium) · Codex CLI",
              "effort": "medium",
              "origin": "reference (hard head-to-head)",
              "strict": "16/16",
              "ci": "81% to 100%",
              "formatMisses": 0,
              "wrong": 0,
              "medianTotal": 13.11,
              "rangeTotal": "8.5 to 61.6",
              "medianOut": 335,
              "meanOut": 530,
              "medianReasoning": 150,
              "perPass": 0.02564
            },
            {
              "config": "GPT-6.1 Sol (high) · Codex CLI",
              "effort": "high",
              "origin": "reference (hard head-to-head)",
              "strict": "16/16",
              "ci": "81% to 100%",
              "formatMisses": 0,
              "wrong": 0,
              "medianTotal": 18.12,
              "rangeTotal": "11.7 to 92.2",
              "medianOut": 436,
              "meanOut": 702,
              "medianReasoning": 225,
              "perPass": 0.01514
            }
          ]
        }
      ],
      "related": [
        "hard-model-head-to-head",
        "caching-consistency"
      ]
    },
    {
      "slug": "caching-consistency",
      "title": "Prompt caching and run-to-run consistency in Claude Code and Codex CLI",
      "seoTitle": "Prompt caching savings and LLM consistency, measured",
      "description": "135 calls: cache hit share and list-price savings in multi-turn sessions, and pass rate and answer diversity over 10 repeats of the same prompt.",
      "question": "When a CLI session reuses a fixed context, how much input comes from the cache, what does that save at list price, and does it change latency? When the same prompt runs 10 times, how much do the pass rate, the answer and the time vary?",
      "answer": "Caching: inside one Claude Code session, turns 2-5 read 97% of their input from the cache on average (turn 1: 19%, the CLI's own prefix). At list price, a calculation, all recorded turns cost Sonnet $0.1350 vs $0.2698 (50% less) and Opus $0.2551 vs $0.5442 (53% less) without the cache. Turn 1 costs more with the cache, because a 1-hour cache write costs twice the input price. A new session did not reuse the cache of an earlier one: on turn 1, all 4 later sessions wrote the ledger to the cache again. The cause was not tested. The cache showed no clear speed effect: median turn time was Sonnet 1.6 s on turn 1 vs 1.6 s on turns 2-5 and Opus 1.9 s on turn 1 vs 2.4 s on turns 2-5, and the fastest-to-slowest ranges overlap. Codex CLI (GPT-6.1 Sol) read 99% of later-turn input from its cache on a larger context; its app-server reports no cache writes, so no cost is calculated for it. Consistency: 7 of 9 model-and-prompt cells passed all 10 repetitions (95% interval 72% to 100%). Haiku passed 0/10 on the exact-number prompt; Haiku passed 1/10 on the JSON prompt (9 more were correct but in the wrong format). Haiku gave the same wrong answer every time (289; expected 282): consistent is not the same as correct. The code-fix prompt gave 6 different code bodies for Haiku, 3 different code bodies for Sonnet and 6 different code bodies for GPT-6.1 Sol (medium).",
      "date": "2026-10-06",
      "updated": "2026-10-06",
      "tags": [
        "prompt-caching",
        "consistency",
        "variance",
        "claude-code",
        "codex-cli",
        "claude-haiku",
        "claude-sonnet",
        "claude-opus",
        "gpt-6-1-sol",
        "calculation"
      ],
      "method": [
        "Protocols declared before the first call, one per route. Every attempt is kept; nothing was retried.",
        "Caching: 9 sessions (3 × Claude Sonnet 5.5 · Claude Code, 3 × Claude Opus 5.5 · Claude Code and 3 × GPT-6.1 Sol (medium) · Codex CLI), 5 turns each. A session is one CLI process. Turn 1 sends a seeded synthetic stock ledger plus question 1; turns 2-5 send one short question each (lookups, a count, an arg-max), each with one exact answer. Each turn is one model request.",
        "A 2-call probe sized the context before the run (not part of any cell). The ledger was larger than declared, so it was cut once, from 170 to 100 lines, and the answers were recomputed, as the protocol allowed.",
        "The Codex sessions ran before that cut, on the 170-line ledger (16,197 characters vs 9,651). The two routes are reported side by side, never as a like-for-like pair.",
        "Cache counters as the provider reports them: Claude Code gives uncached input, cache reads and cache writes (with the 5-minute and 1-hour split); the Codex app-server gives input (cached included) and cached input, and no cache writes.",
        "Cost with and without the cache (Claude only): see the calculation source. Every write in this run was a 1-hour write, so the 5-minute multiplier (an assumption) was not used.",
        "Consistency: 3 prompts (an exact number, a JSON object with exact keys, a small code fix), each with a deterministic validator; 10 repetitions per prompt for Claude Haiku 4.5 · Claude Code, Claude Sonnet 5.5 · Claude Code and GPT-6.1 Sol (medium) · Codex CLI. One-shot calls, one at a time per account. Answer diversity counts distinct normalized answers; the raw extract holds ordinal answer ids, never the text.",
        "Validator controls ran before inference on each route: every reference answer passes, every plausible wrong answer fails, and a wrapped reference is flagged as a format miss.",
        "Isolation as in the hard head-to-head: fresh empty working folder, tools off, no MCP servers, no session persistence across processes, provider-default caching. Claude at its default effort; GPT-6.1 Sol at medium.",
        "No batch stopped early and nothing was trimmed. The Claude CLI reported a rate-limit status of \"allowed_warning\" on 7 of 30 cache turns; no call was refused and the run did not stop."
      ],
      "caveats": [
        "Cache figures come from 3 sessions per model; they describe this CLI version and this context size. A different working folder, prompt order or cache lifetime can change them.",
        "Cross-session reuse did not happen here. The cause was not tested: the CLI may add per-process context before the user message. Do not read it as a provider property.",
        "Costs are list-price calculations; the calls used flat subscriptions.",
        "Consistency rests on 10 repetitions per cell: a 10/10 has a 95% interval of 72% to 100%.",
        "The JSON prompt fails a reply in a code fence even when the JSON is right. The table counts these format misses apart from wrong answers.",
        "Claude rows ran at default effort and GPT-6.1 Sol at medium; Codex CLI adds its own system prompt and tool schemas. Rows across routes compare route + model pairs."
      ],
      "sourceIds": [
        "agent-caching-consistency",
        "calc-cache-pricing",
        "price-anthropic"
      ],
      "stats": [
        {
          "id": "caching-saving-sonnet",
          "label": "List-price saving from the cache over 15 turns, Claude Sonnet 5.5 · Claude Code (calculation)",
          "value": 0.4996,
          "unit": "rate",
          "display": "50% ($0.1350 vs $0.2698)",
          "note": "Calculation from recorded tokens and list prices; not a bill."
        },
        {
          "id": "caching-saving-opus",
          "label": "List-price saving from the cache over 15 turns, Claude Opus 5.5 · Claude Code (calculation)",
          "value": 0.5313,
          "unit": "rate",
          "display": "53% ($0.2551 vs $0.5442)",
          "note": "Calculation from recorded tokens and list prices; not a bill."
        },
        {
          "id": "caching-cross-session-reuse",
          "label": "Later sessions whose first turn read the ledger from an earlier session’s cache",
          "value": 0,
          "unit": "count",
          "display": "0 of 4",
          "n": 4
        },
        {
          "id": "consistency-perfect-cells",
          "label": "Model-and-prompt cells that passed all 10 repetitions",
          "value": 7,
          "unit": "count",
          "display": "7 of 9",
          "n": 9
        },
        {
          "id": "caching-consistency-calls",
          "label": "Calls in this study (every one counted)",
          "value": 135,
          "unit": "calls",
          "display": "135 (45 cache turns, 90 repeated prompts)"
        }
      ],
      "charts": [
        {
          "id": "caching-read-share-by-turn",
          "title": "Share of input read from the cache, by turn in a session",
          "subtitle": "Mean over sessions; turn 1 sends the ledger, turns 2-5 send one short question each",
          "kind": "line",
          "unit": "rate",
          "xLabel": "Turn in the session",
          "yLabel": "Input tokens read from cache",
          "series": [
            {
              "name": "Claude Sonnet 5.5 · Claude Code",
              "points": [
                {
                  "label": "Turn 1",
                  "value": 0.1868,
                  "n": 3
                },
                {
                  "label": "Turn 2",
                  "value": 0.9924,
                  "n": 3
                },
                {
                  "label": "Turn 3",
                  "value": 0.9896,
                  "n": 3
                },
                {
                  "label": "Turn 4",
                  "value": 0.992,
                  "n": 3
                },
                {
                  "label": "Turn 5",
                  "value": 0.9074,
                  "n": 3
                }
              ]
            },
            {
              "name": "Claude Opus 5.5 · Claude Code",
              "points": [
                {
                  "label": "Turn 1",
                  "value": 0.1869,
                  "n": 3
                },
                {
                  "label": "Turn 2",
                  "value": 0.9924,
                  "n": 3
                },
                {
                  "label": "Turn 3",
                  "value": 0.9866,
                  "n": 3
                },
                {
                  "label": "Turn 4",
                  "value": 0.9893,
                  "n": 3
                },
                {
                  "label": "Turn 5",
                  "value": 0.9031,
                  "n": 3
                }
              ]
            },
            {
              "name": "GPT-6.1 Sol (medium) · Codex CLI",
              "points": [
                {
                  "label": "Turn 1",
                  "value": 0.5545,
                  "n": 3
                },
                {
                  "label": "Turn 2",
                  "value": 0.9871,
                  "n": 3
                },
                {
                  "label": "Turn 3",
                  "value": 0.9892,
                  "n": 3
                },
                {
                  "label": "Turn 4",
                  "value": 0.9898,
                  "n": 3
                },
                {
                  "label": "Turn 5",
                  "value": 0.9786,
                  "n": 3
                }
              ]
            }
          ],
          "note": "Claude Code: cache reads ÷ (uncached input + cache reads + cache writes). Codex CLI: cached input ÷ input, as its app-server reports them; it reports no cache writes, and its input includes its own system prompt and tool schemas. The Codex sessions used the ledger before it was cut (16,197 characters vs 9,651), so the two routes are not a like-for-like pair. Measured shares, not pass rates.",
          "sourceIds": [
            "agent-caching-consistency"
          ]
        },
        {
          "id": "caching-cost-with-without",
          "title": "List-price cost of 5-question sessions with and without the cache (calculation)",
          "subtitle": "All recorded turns per model; the same reported tokens priced two ways",
          "kind": "grouped-bar",
          "unit": "usd",
          "yLabel": "USD (list price)",
          "series": [
            {
              "name": "With the cache, as recorded",
              "points": [
                {
                  "label": "Claude Sonnet 5.5 · Claude Code",
                  "value": 0.135003,
                  "n": 15
                },
                {
                  "label": "Claude Opus 5.5 · Claude Code",
                  "value": 0.255057,
                  "n": 15
                }
              ]
            },
            {
              "name": "Without a cache: every input token at the input price",
              "points": [
                {
                  "label": "Claude Sonnet 5.5 · Claude Code",
                  "value": 0.269788,
                  "n": 15
                },
                {
                  "label": "Claude Opus 5.5 · Claude Code",
                  "value": 0.544228,
                  "n": 15
                }
              ]
            }
          ],
          "note": "Calculation, not a bill: the calls ran on a subscription. Cache reads at the cache-read price, 1-hour cache writes at 2× the input price (every write in this run was a 1-hour write). Codex CLI is not priced here: it reports no cache-write count.",
          "sourceIds": [
            "agent-caching-consistency",
            "calc-cache-pricing",
            "price-anthropic"
          ]
        },
        {
          "id": "caching-latency-first-vs-later",
          "title": "Time per turn: first turn vs later turns in a cached session",
          "subtitle": "Median; whiskers = fastest and slowest turn",
          "kind": "dot-range",
          "unit": "seconds",
          "yLabel": "Seconds",
          "series": [
            {
              "name": "Turn 1 (writes the ledger to the cache)",
              "points": [
                {
                  "label": "Claude Sonnet 5.5 · Claude Code",
                  "value": 1.64,
                  "lo": 1.58,
                  "hi": 1.79,
                  "n": 3
                },
                {
                  "label": "Claude Opus 5.5 · Claude Code",
                  "value": 1.9,
                  "lo": 1.78,
                  "hi": 4.36,
                  "n": 3
                }
              ]
            },
            {
              "name": "Turns 2-5 (read the ledger from the cache)",
              "points": [
                {
                  "label": "Claude Sonnet 5.5 · Claude Code",
                  "value": 1.61,
                  "lo": 1.35,
                  "hi": 5.63,
                  "n": 12
                },
                {
                  "label": "Claude Opus 5.5 · Claude Code",
                  "value": 2.4,
                  "lo": 1.63,
                  "hi": 12.67,
                  "n": 12
                }
              ]
            }
          ],
          "note": "Whiskers are a range (fastest and slowest turn), not a confidence interval. Turns ask different questions: the slow later turns are the counting question, which produced the most output.",
          "whisker": "minmax",
          "sourceIds": [
            "agent-caching-consistency"
          ]
        },
        {
          "id": "consistency-pass-rate",
          "title": "Same prompt, 10 times: strict pass rate",
          "subtitle": "One series per prompt; whiskers are 95% Wilson intervals",
          "kind": "dot-range",
          "unit": "rate",
          "yLabel": "Passed",
          "series": [
            {
              "name": "Exact number",
              "points": [
                {
                  "label": "Claude Haiku 4.5 · Claude Code",
                  "value": 0,
                  "lo": 0,
                  "hi": 0.2775,
                  "n": 10
                },
                {
                  "label": "Claude Sonnet 5.5 · Claude Code",
                  "value": 1,
                  "lo": 0.7225,
                  "hi": 1,
                  "n": 10
                },
                {
                  "label": "GPT-6.1 Sol (medium) · Codex CLI",
                  "value": 1,
                  "lo": 0.7225,
                  "hi": 1,
                  "n": 10
                }
              ]
            },
            {
              "name": "JSON object",
              "points": [
                {
                  "label": "Claude Haiku 4.5 · Claude Code",
                  "value": 0.1,
                  "lo": 0.0179,
                  "hi": 0.4042,
                  "n": 10
                },
                {
                  "label": "Claude Sonnet 5.5 · Claude Code",
                  "value": 1,
                  "lo": 0.7225,
                  "hi": 1,
                  "n": 10
                },
                {
                  "label": "GPT-6.1 Sol (medium) · Codex CLI",
                  "value": 1,
                  "lo": 0.7225,
                  "hi": 1,
                  "n": 10
                }
              ]
            },
            {
              "name": "Code fix",
              "points": [
                {
                  "label": "Claude Haiku 4.5 · Claude Code",
                  "value": 1,
                  "lo": 0.7225,
                  "hi": 1,
                  "n": 10
                },
                {
                  "label": "Claude Sonnet 5.5 · Claude Code",
                  "value": 1,
                  "lo": 0.7225,
                  "hi": 1,
                  "n": 10
                },
                {
                  "label": "GPT-6.1 Sol (medium) · Codex CLI",
                  "value": 1,
                  "lo": 0.7225,
                  "hi": 1,
                  "n": 10
                }
              ]
            }
          ],
          "note": "Whiskers are 95% Wilson intervals over 10 repetitions. Strict: the whole reply passes the validator. A correct answer in the wrong format (for example in a code fence) is a format miss and never a pass; the table lists format misses apart from wrong answers.",
          "whisker": "ci95",
          "sourceIds": [
            "agent-caching-consistency"
          ]
        },
        {
          "id": "consistency-distinct-answers",
          "title": "Same prompt, 10 times: how many different answers",
          "subtitle": "Distinct normalized answers over 10 repetitions (1 = the same answer every time)",
          "kind": "grouped-bar",
          "unit": "count",
          "yLabel": "Distinct answers",
          "series": [
            {
              "name": "Exact number",
              "points": [
                {
                  "label": "Claude Haiku 4.5 · Claude Code",
                  "value": 1,
                  "n": 10
                },
                {
                  "label": "Claude Sonnet 5.5 · Claude Code",
                  "value": 1,
                  "n": 10
                },
                {
                  "label": "GPT-6.1 Sol (medium) · Codex CLI",
                  "value": 1,
                  "n": 10
                }
              ]
            },
            {
              "name": "JSON object",
              "points": [
                {
                  "label": "Claude Haiku 4.5 · Claude Code",
                  "value": 1,
                  "n": 10
                },
                {
                  "label": "Claude Sonnet 5.5 · Claude Code",
                  "value": 1,
                  "n": 10
                },
                {
                  "label": "GPT-6.1 Sol (medium) · Codex CLI",
                  "value": 1,
                  "n": 10
                }
              ]
            },
            {
              "name": "Code fix",
              "points": [
                {
                  "label": "Claude Haiku 4.5 · Claude Code",
                  "value": 6,
                  "n": 10
                },
                {
                  "label": "Claude Sonnet 5.5 · Claude Code",
                  "value": 3,
                  "n": 10
                },
                {
                  "label": "GPT-6.1 Sol (medium) · Codex CLI",
                  "value": 6,
                  "n": 10
                }
              ]
            }
          ],
          "note": "Normalized: the last number for the exact-number prompt; key-sorted JSON for the JSON prompt; for the code fix, the code without fences, comments, whitespace, quote style and line-end semicolons. One distinct answer is not the same as a correct answer: an answer can be the same and wrong every time. Different code can be equally correct.",
          "sourceIds": [
            "agent-caching-consistency"
          ]
        },
        {
          "id": "consistency-latency-spread",
          "title": "Same prompt, 10 times: time per call",
          "subtitle": "Median; whiskers = fastest and slowest of 10 calls",
          "kind": "dot-range",
          "unit": "seconds",
          "yLabel": "Seconds",
          "series": [
            {
              "name": "Exact number",
              "points": [
                {
                  "label": "Claude Haiku 4.5 · Claude Code",
                  "value": 5.06,
                  "lo": 4.42,
                  "hi": 6.2,
                  "n": 10
                },
                {
                  "label": "Claude Sonnet 5.5 · Claude Code",
                  "value": 6.89,
                  "lo": 5.81,
                  "hi": 7.81,
                  "n": 10
                },
                {
                  "label": "GPT-6.1 Sol (medium) · Codex CLI",
                  "value": 13.38,
                  "lo": 12.29,
                  "hi": 17.97,
                  "n": 10
                }
              ]
            },
            {
              "name": "JSON object",
              "points": [
                {
                  "label": "Claude Haiku 4.5 · Claude Code",
                  "value": 7.03,
                  "lo": 5.28,
                  "hi": 12.27,
                  "n": 10
                },
                {
                  "label": "Claude Sonnet 5.5 · Claude Code",
                  "value": 2.89,
                  "lo": 2.68,
                  "hi": 5.3,
                  "n": 10
                },
                {
                  "label": "GPT-6.1 Sol (medium) · Codex CLI",
                  "value": 6.42,
                  "lo": 5.25,
                  "hi": 8.26,
                  "n": 10
                }
              ]
            },
            {
              "name": "Code fix",
              "points": [
                {
                  "label": "Claude Haiku 4.5 · Claude Code",
                  "value": 5.95,
                  "lo": 4.89,
                  "hi": 7.33,
                  "n": 10
                },
                {
                  "label": "Claude Sonnet 5.5 · Claude Code",
                  "value": 2.67,
                  "lo": 2.32,
                  "hi": 4.34,
                  "n": 10
                },
                {
                  "label": "GPT-6.1 Sol (medium) · Codex CLI",
                  "value": 11.29,
                  "lo": 9.08,
                  "hi": 14.85,
                  "n": 10
                }
              ]
            }
          ],
          "note": "Whiskers are a range (fastest and slowest call), not a confidence interval. The interquartile range and the coefficient of variation are in the table. Codex CLI timings include its start-up and its larger system prompt.",
          "whisker": "minmax",
          "sourceIds": [
            "agent-caching-consistency"
          ]
        }
      ],
      "tables": [
        {
          "id": "caching-turns",
          "title": "Cache counters per turn (mean over sessions)",
          "columns": [
            {
              "key": "config",
              "label": "Configuration",
              "unit": "text"
            },
            {
              "key": "turn",
              "label": "Turn",
              "unit": "count"
            },
            {
              "key": "sessions",
              "label": "Sessions",
              "unit": "count"
            },
            {
              "key": "correct",
              "label": "Exact answers",
              "unit": "text"
            },
            {
              "key": "input",
              "label": "Input tokens (all)",
              "unit": "tokens"
            },
            {
              "key": "read",
              "label": "Read from cache",
              "unit": "tokens"
            },
            {
              "key": "written",
              "label": "Written to cache",
              "unit": "text"
            },
            {
              "key": "share",
              "label": "Read share",
              "unit": "rate"
            },
            {
              "key": "medianTotal",
              "label": "Median time (s)",
              "unit": "seconds"
            },
            {
              "key": "withCache",
              "label": "USD with cache (calculation)",
              "unit": "usd"
            },
            {
              "key": "without",
              "label": "USD without cache (calculation)",
              "unit": "usd"
            }
          ],
          "rows": [
            {
              "config": "Claude Sonnet 5.5 · Claude Code",
              "turn": 1,
              "sessions": 3,
              "correct": "3/3",
              "input": 7831,
              "read": 1463,
              "written": "6,366",
              "share": 0.1868,
              "medianTotal": 1.64,
              "withCache": 0.025792,
              "without": 0.015693
            },
            {
              "config": "Claude Sonnet 5.5 · Claude Code",
              "turn": 2,
              "sessions": 3,
              "correct": "3/3",
              "input": 7889,
              "read": 7829,
              "written": "58",
              "share": 0.9924,
              "medianTotal": 1.54,
              "withCache": 0.001862,
              "without": 0.015839
            },
            {
              "config": "Claude Sonnet 5.5 · Claude Code",
              "turn": 3,
              "sessions": 3,
              "correct": "3/3",
              "input": 7970,
              "read": 7887,
              "written": "81",
              "share": 0.9896,
              "medianTotal": 1.53,
              "withCache": 0.001955,
              "without": 0.015991
            },
            {
              "config": "Claude Sonnet 5.5 · Claude Code",
              "turn": 4,
              "sessions": 3,
              "correct": "3/3",
              "input": 8032,
              "read": 7968,
              "written": "62",
              "share": 0.992,
              "medianTotal": 5.2,
              "withCache": 0.009576,
              "without": 0.023795
            },
            {
              "config": "Claude Sonnet 5.5 · Claude Code",
              "turn": 5,
              "sessions": 3,
              "correct": "3/3",
              "input": 8861,
              "read": 8030,
              "written": "829",
              "share": 0.9074,
              "medianTotal": 1.77,
              "withCache": 0.005816,
              "without": 0.018613
            },
            {
              "config": "Claude Opus 5.5 · Claude Code",
              "turn": 1,
              "sessions": 3,
              "correct": "3/3",
              "input": 7828,
              "read": 1463,
              "written": "6,363",
              "share": 0.1869,
              "medianTotal": 1.9,
              "withCache": 0.051265,
              "without": 0.031372
            },
            {
              "config": "Claude Opus 5.5 · Claude Code",
              "turn": 2,
              "sessions": 3,
              "correct": "3/3",
              "input": 7886,
              "read": 7826,
              "written": "58",
              "share": 0.9924,
              "medianTotal": 1.63,
              "withCache": 0.002637,
              "without": 0.032144
            },
            {
              "config": "Claude Opus 5.5 · Claude Code",
              "turn": 3,
              "sessions": 3,
              "correct": "3/3",
              "input": 7991,
              "read": 7884,
              "written": "105",
              "share": 0.9866,
              "medianTotal": 1.81,
              "withCache": 0.002965,
              "without": 0.032504
            },
            {
              "config": "Claude Opus 5.5 · Claude Code",
              "turn": 4,
              "sessions": 3,
              "correct": "3/3",
              "input": 8075,
              "read": 7989,
              "written": "84",
              "share": 0.9893,
              "medianTotal": 6.09,
              "withCache": 0.018471,
              "without": 0.048493
            },
            {
              "config": "Claude Opus 5.5 · Claude Code",
              "turn": 5,
              "sessions": 3,
              "correct": "3/3",
              "input": 8941,
              "read": 8073,
              "written": "866",
              "share": 0.9031,
              "medianTotal": 2.09,
              "withCache": 0.009681,
              "without": 0.036896
            },
            {
              "config": "GPT-6.1 Sol (medium) · Codex CLI",
              "turn": 1,
              "sessions": 3,
              "correct": "3/3",
              "input": 18774,
              "read": 10411,
              "written": "not reported",
              "share": 0.5545,
              "medianTotal": 3.56,
              "withCache": null,
              "without": null
            },
            {
              "config": "GPT-6.1 Sol (medium) · Codex CLI",
              "turn": 2,
              "sessions": 3,
              "correct": "3/3",
              "input": 18803,
              "read": 18560,
              "written": "not reported",
              "share": 0.9871,
              "medianTotal": 2.39,
              "withCache": null,
              "without": null
            },
            {
              "config": "GPT-6.1 Sol (medium) · Codex CLI",
              "turn": 3,
              "sessions": 3,
              "correct": "3/3",
              "input": 18848,
              "read": 18645,
              "written": "not reported",
              "share": 0.9892,
              "medianTotal": 3.11,
              "withCache": null,
              "without": null
            },
            {
              "config": "GPT-6.1 Sol (medium) · Codex CLI",
              "turn": 4,
              "sessions": 3,
              "correct": "3/3",
              "input": 18880,
              "read": 18688,
              "written": "not reported",
              "share": 0.9898,
              "medianTotal": 5.97,
              "withCache": null,
              "without": null
            },
            {
              "config": "GPT-6.1 Sol (medium) · Codex CLI",
              "turn": 5,
              "sessions": 3,
              "correct": "3/3",
              "input": 19097,
              "read": 18688,
              "written": "not reported",
              "share": 0.9786,
              "medianTotal": 2.38,
              "withCache": null,
              "without": null
            }
          ]
        },
        {
          "id": "consistency-cells",
          "title": "Every consistency cell (10 repetitions each)",
          "columns": [
            {
              "key": "config",
              "label": "Configuration",
              "unit": "text"
            },
            {
              "key": "prompt",
              "label": "Prompt",
              "unit": "text"
            },
            {
              "key": "strict",
              "label": "Strict passes",
              "unit": "text"
            },
            {
              "key": "ci",
              "label": "95% interval",
              "unit": "text"
            },
            {
              "key": "formatMisses",
              "label": "Format misses",
              "unit": "count"
            },
            {
              "key": "wrong",
              "label": "Wrong answers",
              "unit": "count"
            },
            {
              "key": "distinct",
              "label": "Distinct answers",
              "unit": "count"
            },
            {
              "key": "distinctRaw",
              "label": "Distinct raw replies",
              "unit": "count"
            },
            {
              "key": "medianTotal",
              "label": "Median time (s)",
              "unit": "seconds"
            },
            {
              "key": "iqr",
              "label": "Interquartile range (s)",
              "unit": "seconds"
            },
            {
              "key": "cv",
              "label": "Coefficient of variation",
              "unit": "score"
            }
          ],
          "rows": [
            {
              "config": "Claude Haiku 4.5 · Claude Code",
              "prompt": "Exact number",
              "strict": "0/10",
              "ci": "0% to 28%",
              "formatMisses": 0,
              "wrong": 10,
              "distinct": 1,
              "distinctRaw": 10,
              "medianTotal": 5.06,
              "iqr": 1.12,
              "cv": 0.13
            },
            {
              "config": "Claude Haiku 4.5 · Claude Code",
              "prompt": "JSON object",
              "strict": "1/10",
              "ci": "2% to 40%",
              "formatMisses": 9,
              "wrong": 0,
              "distinct": 1,
              "distinctRaw": 3,
              "medianTotal": 7.03,
              "iqr": 2.21,
              "cv": 0.28
            },
            {
              "config": "Claude Haiku 4.5 · Claude Code",
              "prompt": "Code fix",
              "strict": "10/10",
              "ci": "72% to 100%",
              "formatMisses": 0,
              "wrong": 0,
              "distinct": 6,
              "distinctRaw": 6,
              "medianTotal": 5.95,
              "iqr": 1.09,
              "cv": 0.13
            },
            {
              "config": "Claude Sonnet 5.5 · Claude Code",
              "prompt": "Exact number",
              "strict": "10/10",
              "ci": "72% to 100%",
              "formatMisses": 0,
              "wrong": 0,
              "distinct": 1,
              "distinctRaw": 1,
              "medianTotal": 6.89,
              "iqr": 0.95,
              "cv": 0.09
            },
            {
              "config": "Claude Sonnet 5.5 · Claude Code",
              "prompt": "JSON object",
              "strict": "10/10",
              "ci": "72% to 100%",
              "formatMisses": 0,
              "wrong": 0,
              "distinct": 1,
              "distinctRaw": 1,
              "medianTotal": 2.89,
              "iqr": 0.54,
              "cv": 0.28
            },
            {
              "config": "Claude Sonnet 5.5 · Claude Code",
              "prompt": "Code fix",
              "strict": "10/10",
              "ci": "72% to 100%",
              "formatMisses": 0,
              "wrong": 0,
              "distinct": 3,
              "distinctRaw": 3,
              "medianTotal": 2.67,
              "iqr": 1.17,
              "cv": 0.26
            },
            {
              "config": "GPT-6.1 Sol (medium) · Codex CLI",
              "prompt": "Exact number",
              "strict": "10/10",
              "ci": "72% to 100%",
              "formatMisses": 0,
              "wrong": 0,
              "distinct": 1,
              "distinctRaw": 1,
              "medianTotal": 13.38,
              "iqr": 1.53,
              "cv": 0.14
            },
            {
              "config": "GPT-6.1 Sol (medium) · Codex CLI",
              "prompt": "JSON object",
              "strict": "10/10",
              "ci": "72% to 100%",
              "formatMisses": 0,
              "wrong": 0,
              "distinct": 1,
              "distinctRaw": 1,
              "medianTotal": 6.42,
              "iqr": 1.8,
              "cv": 0.17
            },
            {
              "config": "GPT-6.1 Sol (medium) · Codex CLI",
              "prompt": "Code fix",
              "strict": "10/10",
              "ci": "72% to 100%",
              "formatMisses": 0,
              "wrong": 0,
              "distinct": 6,
              "distinctRaw": 6,
              "medianTotal": 11.29,
              "iqr": 2.25,
              "cv": 0.15
            }
          ]
        }
      ],
      "related": [
        "effort-ladder",
        "cli-model-latency-tokens"
      ]
    },
    {
      "slug": "agent-memory",
      "title": "Does memory help Claude Code? 8 kinds of agent memory, tested",
      "seoTitle": "Does CLAUDE.md help? Agent memory tested on Claude Code",
      "description": "200 graded Claude Code sessions: no memory, /init, curated, raw notes, dreamed notes, a long handbook, a Stop hook. What helped and what it cost.",
      "question": "Does project memory make Claude Code do better work, and which kind of memory?",
      "answer": "Claude Sonnet 5.5 followed almost every rule it could see without any memory (95% of code rules, 100% of folder rules), but only 6/15 of the team-knowledge checks; with an 11-line curated file it passed 15/15. When a memory file held the late-fee rate, Sonnet used it 15/15 times, even from raw notes that also held a stale rate; without it, Sonnet stopped and asked 7 of 9 times, while Haiku invented a rate and flagged it in 0/6 sessions. The smaller Haiku 4.5 was misled by messy memory: with the raw notes it ran the stale test command in 10/10 sessions, and after one dreaming pass in 1/10. A Stop hook enforced every code rule but could not carry a fact, and it used 1.6× the input tokens of no memory. Full-pass intervals overlap for most pairs (n = 15 per condition).",
      "date": "2026-10-06",
      "updated": "2026-10-06",
      "tags": [
        "claude-code",
        "agent-memory",
        "claude-md",
        "hooks",
        "context-engineering"
      ],
      "method": [
        "One small Node.js repository with real conventions: money in integer cents, coded errors, a clock helper, a log helper, append-only migrations, a generated report, a changelog rule and a late-fee rate that exists only in memory.",
        "Five tasks: refunds, a report formatting bug, an optional phone field, a late fee \"at our standard rate\", and a control task where memory does not matter.",
        "Eight conditions: no memory, the real /init output, a curated 11-line file, 56 raw dated notes (repeats, one-off events and notes that a later note contradicts), the same notes after one \"dreaming\" pass, the curated facts inside a 210-line handbook, a Stop hook that blocks finishing while a rule is broken, and curated plus hook.",
        "Claude Code 2.1.286 headless; Sonnet 3 repetitions per cell, Haiku 2; tools Bash, Read, Edit, Write, Glob, Grep; OS sandbox; the tester's own instruction files excluded and auto memory off (both checked with a probe).",
        "Grading after the session: hidden tests run in a sandbox, plus deterministic checks on the lines the agent added. Full pass = tests and every applicable check. Every attempt counts.",
        "The protocol was declared before the first counted session. One amendment, made before any Haiku session, added the Haiku lane. One erratum: the raw notes file holds 56 notes, not 58 as first written. The frozen handbook has 210 lines, not 211 as the protocol first stated. Measurements did not change."
      ],
      "caveats": [
        "One small synthetic repository and one author of the facts: the curated file is an upper bound written with knowledge of the tasks.",
        "The hook checks the same code rules as the grader. It shows what rules written as code can do; it cannot carry a fact such as the late-fee rate.",
        "n = 15 per condition for Sonnet and 10 for Haiku: most full-pass intervals overlap, so most differences between conditions are not clear.",
        "Costs are the CLI's list-price estimates for subscription sessions, not invoices.",
        "In headless mode an agent that asks a question cannot get an answer, so asking counts as a failure here. In a live session the person would answer; the cost is the round trip."
      ],
      "sourceIds": [
        "agent-memory-study"
      ],
      "stats": [
        {
          "id": "memory-sessions",
          "label": "Claude Code sessions, every one graded (120 Sonnet 5.5, 80 Haiku 4.5)",
          "value": 200,
          "unit": "count",
          "display": "200",
          "n": 200,
          "note": "0 planned sessions not run."
        },
        {
          "id": "memory-team-knowledge-none",
          "label": "Team-knowledge checks passed with no memory (Sonnet 5.5)",
          "value": 0.4,
          "unit": "rate",
          "display": "40% (6/15)",
          "n": 15,
          "ci": [
            0.1982,
            0.6425
          ]
        },
        {
          "id": "memory-team-knowledge-curated",
          "label": "Team-knowledge checks passed with an 11-line curated file (Sonnet 5.5)",
          "value": 1,
          "unit": "rate",
          "display": "100% (15/15)",
          "n": 15,
          "ci": [
            0.7961,
            1
          ]
        },
        {
          "id": "memory-late-fee-known",
          "label": "Late fee right when any memory file held the rate (Sonnet 5.5)",
          "value": 1,
          "unit": "rate",
          "display": "15/15",
          "n": 15,
          "ci": [
            0.7961,
            1
          ]
        },
        {
          "id": "memory-late-fee-asked",
          "label": "Without the rate, sessions that asked for it and wrote no code (Sonnet 5.5)",
          "value": 0.7778,
          "unit": "rate",
          "display": "7/9",
          "n": 9,
          "note": "The other 2 guessed a rate and said it was a placeholder."
        },
        {
          "id": "memory-late-fee-flagged-sonnet",
          "label": "Without the rate: sessions whose final message said the rate was unknown or a guess (Sonnet 5.5)",
          "value": 1,
          "unit": "rate",
          "display": "9/9",
          "n": 9
        },
        {
          "id": "memory-late-fee-flagged-haiku",
          "label": "Without the rate: sessions whose final message said the rate was unknown or a guess (Haiku 4.5)",
          "value": 0,
          "unit": "rate",
          "display": "0/6",
          "n": 6,
          "note": "Every other Haiku session invented a rate and reported the task done."
        },
        {
          "id": "memory-init-broken-command",
          "label": "Sessions with the /init file that ran the broken test command it copied from the README (Sonnet 5.5)",
          "value": 0.8667,
          "unit": "rate",
          "display": "13/15",
          "n": 15
        },
        {
          "id": "memory-hook-input-ratio",
          "label": "Median input tokens per session, Stop hook only vs no memory (calculation, Sonnet 5.5)",
          "value": 1.6,
          "unit": "ratio",
          "display": "1.6×",
          "n": 15
        },
        {
          "id": "memory-haiku-raw-broken-command",
          "label": "Haiku 4.5 with the raw notes: sessions that ran the stale test command",
          "value": 1,
          "unit": "rate",
          "display": "10/10",
          "n": 10
        },
        {
          "id": "memory-haiku-dreamed-broken-command",
          "label": "Haiku 4.5 with the dreamed notes: sessions that ran the stale test command",
          "value": 0.1,
          "unit": "rate",
          "display": "1/10",
          "n": 10
        },
        {
          "id": "memory-total-cost",
          "label": "List-price estimate of every session (calculation, not an invoice)",
          "value": 17.49,
          "unit": "usd",
          "display": "$17.49",
          "n": 200
        }
      ],
      "charts": [
        {
          "id": "memory-full-pass",
          "title": "Full pass rate by kind of memory",
          "subtitle": "Hidden tests pass and every convention check passes · 95% Wilson intervals",
          "kind": "dot-range",
          "unit": "rate",
          "whisker": "ci95",
          "series": [
            {
              "name": "Claude Sonnet 5.5",
              "points": [
                {
                  "label": "No memory",
                  "value": 0.6,
                  "lo": 0.3575,
                  "hi": 0.8018,
                  "n": 15
                },
                {
                  "label": "/init CLAUDE.md",
                  "value": 0.6,
                  "lo": 0.3575,
                  "hi": 0.8018,
                  "n": 15
                },
                {
                  "label": "Curated, 11 lines",
                  "value": 1,
                  "lo": 0.7961,
                  "hi": 1,
                  "n": 15,
                  "highlight": true
                },
                {
                  "label": "Raw notes, 60 lines",
                  "value": 0.9333,
                  "lo": 0.7018,
                  "hi": 0.9881,
                  "n": 15
                },
                {
                  "label": "Dreamed notes",
                  "value": 1,
                  "lo": 0.7961,
                  "hi": 1,
                  "n": 15
                },
                {
                  "label": "Handbook, 210 lines",
                  "value": 1,
                  "lo": 0.7961,
                  "hi": 1,
                  "n": 15
                },
                {
                  "label": "Stop hook only",
                  "value": 0.8,
                  "lo": 0.5481,
                  "hi": 0.9295,
                  "n": 15
                },
                {
                  "label": "Curated + hook",
                  "value": 1,
                  "lo": 0.7961,
                  "hi": 1,
                  "n": 15
                }
              ]
            },
            {
              "name": "Claude Haiku 4.5",
              "points": [
                {
                  "label": "No memory",
                  "value": 0.2,
                  "lo": 0.0567,
                  "hi": 0.5098,
                  "n": 10
                },
                {
                  "label": "/init CLAUDE.md",
                  "value": 0.2,
                  "lo": 0.0567,
                  "hi": 0.5098,
                  "n": 10
                },
                {
                  "label": "Curated, 11 lines",
                  "value": 0.7,
                  "lo": 0.3968,
                  "hi": 0.8922,
                  "n": 10
                },
                {
                  "label": "Raw notes, 60 lines",
                  "value": 0.6,
                  "lo": 0.3127,
                  "hi": 0.8318,
                  "n": 10
                },
                {
                  "label": "Dreamed notes",
                  "value": 0.7,
                  "lo": 0.3968,
                  "hi": 0.8922,
                  "n": 10
                },
                {
                  "label": "Handbook, 210 lines",
                  "value": 0.3,
                  "lo": 0.1078,
                  "hi": 0.6032,
                  "n": 10
                },
                {
                  "label": "Stop hook only",
                  "value": 0.8,
                  "lo": 0.4902,
                  "hi": 0.9433,
                  "n": 10
                },
                {
                  "label": "Curated + hook",
                  "value": 0.9,
                  "lo": 0.5958,
                  "hi": 0.9821,
                  "n": 10
                }
              ]
            }
          ],
          "note": "Claude Code 2.1.286, 5 tasks in one small repository. Sonnet: 3 repetitions per cell (n = 15 per condition); Haiku: 2 (n = 10). A condition is better only when its interval does not overlap the other's.",
          "sourceIds": [
            "agent-memory-study"
          ]
        },
        {
          "id": "memory-knowledge-class",
          "title": "Where memory helps: what the repo shows vs what only the team knows",
          "subtitle": "Share of convention checks passed, pooled by kind of knowledge (Claude Sonnet 5.5)",
          "kind": "grouped-bar",
          "unit": "rate",
          "series": [
            {
              "name": "Rules the code already shows",
              "points": [
                {
                  "label": "No memory",
                  "value": 0.95,
                  "lo": 0.863,
                  "hi": 0.9829,
                  "n": 60
                },
                {
                  "label": "/init CLAUDE.md",
                  "value": 0.95,
                  "lo": 0.863,
                  "hi": 0.9829,
                  "n": 60
                },
                {
                  "label": "Curated, 11 lines",
                  "value": 1,
                  "lo": 0.9398,
                  "hi": 1,
                  "n": 60
                },
                {
                  "label": "Raw notes, 60 lines",
                  "value": 1,
                  "lo": 0.9398,
                  "hi": 1,
                  "n": 60
                },
                {
                  "label": "Dreamed notes",
                  "value": 1,
                  "lo": 0.9398,
                  "hi": 1,
                  "n": 60
                },
                {
                  "label": "Handbook, 210 lines",
                  "value": 1,
                  "lo": 0.9398,
                  "hi": 1,
                  "n": 60
                },
                {
                  "label": "Stop hook only",
                  "value": 1,
                  "lo": 0.9398,
                  "hi": 1,
                  "n": 60
                },
                {
                  "label": "Curated + hook",
                  "value": 1,
                  "lo": 0.9398,
                  "hi": 1,
                  "n": 60
                }
              ]
            },
            {
              "name": "Rules the folders hint at",
              "points": [
                {
                  "label": "No memory",
                  "value": 1,
                  "lo": 0.8454,
                  "hi": 1,
                  "n": 21
                },
                {
                  "label": "/init CLAUDE.md",
                  "value": 1,
                  "lo": 0.8454,
                  "hi": 1,
                  "n": 21
                },
                {
                  "label": "Curated, 11 lines",
                  "value": 1,
                  "lo": 0.8454,
                  "hi": 1,
                  "n": 21
                },
                {
                  "label": "Raw notes, 60 lines",
                  "value": 1,
                  "lo": 0.8454,
                  "hi": 1,
                  "n": 21
                },
                {
                  "label": "Dreamed notes",
                  "value": 1,
                  "lo": 0.8454,
                  "hi": 1,
                  "n": 21
                },
                {
                  "label": "Handbook, 210 lines",
                  "value": 1,
                  "lo": 0.8454,
                  "hi": 1,
                  "n": 21
                },
                {
                  "label": "Stop hook only",
                  "value": 1,
                  "lo": 0.8454,
                  "hi": 1,
                  "n": 21
                },
                {
                  "label": "Curated + hook",
                  "value": 1,
                  "lo": 0.8454,
                  "hi": 1,
                  "n": 21
                }
              ]
            },
            {
              "name": "Team knowledge only",
              "points": [
                {
                  "label": "No memory",
                  "value": 0.4,
                  "lo": 0.1982,
                  "hi": 0.6425,
                  "n": 15
                },
                {
                  "label": "/init CLAUDE.md",
                  "value": 0.6667,
                  "lo": 0.4171,
                  "hi": 0.8482,
                  "n": 15
                },
                {
                  "label": "Curated, 11 lines",
                  "value": 1,
                  "lo": 0.7961,
                  "hi": 1,
                  "n": 15
                },
                {
                  "label": "Raw notes, 60 lines",
                  "value": 1,
                  "lo": 0.7961,
                  "hi": 1,
                  "n": 15
                },
                {
                  "label": "Dreamed notes",
                  "value": 1,
                  "lo": 0.7961,
                  "hi": 1,
                  "n": 15
                },
                {
                  "label": "Handbook, 210 lines",
                  "value": 1,
                  "lo": 0.7961,
                  "hi": 1,
                  "n": 15
                },
                {
                  "label": "Stop hook only",
                  "value": 0.6667,
                  "lo": 0.4171,
                  "hi": 0.8482,
                  "n": 15
                },
                {
                  "label": "Curated + hook",
                  "value": 1,
                  "lo": 0.7961,
                  "hi": 1,
                  "n": 15
                }
              ]
            }
          ],
          "note": "Code shows: money in cents, coded errors, the clock helper, the log helper. Folders hint: append-only migrations, a new migration, the generated report. Team knowledge only: the changelog rule and the late-fee rate.",
          "sourceIds": [
            "agent-memory-study"
          ]
        },
        {
          "id": "memory-team-knowledge-by-model",
          "title": "Team knowledge followed, Sonnet vs Haiku",
          "subtitle": "Changelog rule and late-fee rate, pooled · 95% Wilson intervals",
          "kind": "dot-range",
          "unit": "rate",
          "whisker": "ci95",
          "series": [
            {
              "name": "Claude Sonnet 5.5",
              "points": [
                {
                  "label": "No memory",
                  "value": 0.4,
                  "lo": 0.1982,
                  "hi": 0.6425,
                  "n": 15
                },
                {
                  "label": "/init CLAUDE.md",
                  "value": 0.6667,
                  "lo": 0.4171,
                  "hi": 0.8482,
                  "n": 15
                },
                {
                  "label": "Curated, 11 lines",
                  "value": 1,
                  "lo": 0.7961,
                  "hi": 1,
                  "n": 15
                },
                {
                  "label": "Raw notes, 60 lines",
                  "value": 1,
                  "lo": 0.7961,
                  "hi": 1,
                  "n": 15
                },
                {
                  "label": "Dreamed notes",
                  "value": 1,
                  "lo": 0.7961,
                  "hi": 1,
                  "n": 15
                },
                {
                  "label": "Handbook, 210 lines",
                  "value": 1,
                  "lo": 0.7961,
                  "hi": 1,
                  "n": 15
                },
                {
                  "label": "Stop hook only",
                  "value": 0.6667,
                  "lo": 0.4171,
                  "hi": 0.8482,
                  "n": 15
                },
                {
                  "label": "Curated + hook",
                  "value": 1,
                  "lo": 0.7961,
                  "hi": 1,
                  "n": 15
                }
              ]
            },
            {
              "name": "Claude Haiku 4.5",
              "points": [
                {
                  "label": "No memory",
                  "value": 0,
                  "lo": 0,
                  "hi": 0.2775,
                  "n": 10
                },
                {
                  "label": "/init CLAUDE.md",
                  "value": 0.1,
                  "lo": 0.0179,
                  "hi": 0.4042,
                  "n": 10
                },
                {
                  "label": "Curated, 11 lines",
                  "value": 0.8,
                  "lo": 0.4902,
                  "hi": 0.9433,
                  "n": 10
                },
                {
                  "label": "Raw notes, 60 lines",
                  "value": 0.6,
                  "lo": 0.3127,
                  "hi": 0.8318,
                  "n": 10
                },
                {
                  "label": "Dreamed notes",
                  "value": 0.8,
                  "lo": 0.4902,
                  "hi": 0.9433,
                  "n": 10
                },
                {
                  "label": "Handbook, 210 lines",
                  "value": 0.3,
                  "lo": 0.1078,
                  "hi": 0.6032,
                  "n": 10
                },
                {
                  "label": "Stop hook only",
                  "value": 0.8,
                  "lo": 0.4902,
                  "hi": 0.9433,
                  "n": 10
                },
                {
                  "label": "Curated + hook",
                  "value": 1,
                  "lo": 0.7225,
                  "hi": 1,
                  "n": 10
                }
              ]
            }
          ],
          "note": "Both models had the same memory files. The smaller model followed the team rules less often when the facts sat in long or messy files.",
          "sourceIds": [
            "agent-memory-study"
          ]
        },
        {
          "id": "memory-late-fee",
          "title": "\"Charge our standard late fee\": what Sonnet 5.5 did",
          "subtitle": "The rate (1.25%, decided 2026-09-15) is in no file of the repository; the raw notes also hold a stale 2%",
          "kind": "stacked-bar",
          "unit": "count",
          "series": [
            {
              "name": "Used the current rate (1.25%)",
              "points": [
                {
                  "label": "No memory",
                  "value": 0,
                  "n": 3
                },
                {
                  "label": "/init CLAUDE.md",
                  "value": 0,
                  "n": 3
                },
                {
                  "label": "Curated, 11 lines",
                  "value": 3,
                  "n": 3,
                  "highlight": true
                },
                {
                  "label": "Raw notes, 60 lines",
                  "value": 3,
                  "n": 3,
                  "highlight": true
                },
                {
                  "label": "Dreamed notes",
                  "value": 3,
                  "n": 3,
                  "highlight": true
                },
                {
                  "label": "Handbook, 210 lines",
                  "value": 3,
                  "n": 3,
                  "highlight": true
                },
                {
                  "label": "Stop hook only",
                  "value": 0,
                  "n": 3
                },
                {
                  "label": "Curated + hook",
                  "value": 3,
                  "n": 3,
                  "highlight": true
                }
              ]
            },
            {
              "name": "Asked for the rate, wrote no code",
              "points": [
                {
                  "label": "No memory",
                  "value": 3,
                  "n": 3
                },
                {
                  "label": "/init CLAUDE.md",
                  "value": 2,
                  "n": 3
                },
                {
                  "label": "Curated, 11 lines",
                  "value": 0,
                  "n": 3
                },
                {
                  "label": "Raw notes, 60 lines",
                  "value": 0,
                  "n": 3
                },
                {
                  "label": "Dreamed notes",
                  "value": 0,
                  "n": 3
                },
                {
                  "label": "Handbook, 210 lines",
                  "value": 0,
                  "n": 3
                },
                {
                  "label": "Stop hook only",
                  "value": 2,
                  "n": 3
                },
                {
                  "label": "Curated + hook",
                  "value": 0,
                  "n": 3
                }
              ]
            },
            {
              "name": "Guessed another rate",
              "points": [
                {
                  "label": "No memory",
                  "value": 0,
                  "n": 3
                },
                {
                  "label": "/init CLAUDE.md",
                  "value": 1,
                  "n": 3
                },
                {
                  "label": "Curated, 11 lines",
                  "value": 0,
                  "n": 3
                },
                {
                  "label": "Raw notes, 60 lines",
                  "value": 0,
                  "n": 3
                },
                {
                  "label": "Dreamed notes",
                  "value": 0,
                  "n": 3
                },
                {
                  "label": "Handbook, 210 lines",
                  "value": 0,
                  "n": 3
                },
                {
                  "label": "Stop hook only",
                  "value": 1,
                  "n": 3
                },
                {
                  "label": "Curated + hook",
                  "value": 0,
                  "n": 3
                }
              ]
            }
          ],
          "note": "Late-fee task, 3 sessions per condition. \"Asked\" means the agent searched the repository, found no rate and stopped with a question instead of code.",
          "sourceIds": [
            "agent-memory-study"
          ]
        },
        {
          "id": "memory-late-fee-haiku",
          "title": "\"Charge our standard late fee\": what Haiku 4.5 did",
          "subtitle": "The rate (1.25%, decided 2026-09-15) is in no file of the repository; the raw notes also hold a stale 2%",
          "kind": "stacked-bar",
          "unit": "count",
          "series": [
            {
              "name": "Used the current rate (1.25%)",
              "points": [
                {
                  "label": "No memory",
                  "value": 0,
                  "n": 2
                },
                {
                  "label": "/init CLAUDE.md",
                  "value": 0,
                  "n": 2
                },
                {
                  "label": "Curated, 11 lines",
                  "value": 2,
                  "n": 2,
                  "highlight": true
                },
                {
                  "label": "Raw notes, 60 lines",
                  "value": 2,
                  "n": 2,
                  "highlight": true
                },
                {
                  "label": "Dreamed notes",
                  "value": 2,
                  "n": 2,
                  "highlight": true
                },
                {
                  "label": "Handbook, 210 lines",
                  "value": 2,
                  "n": 2,
                  "highlight": true
                },
                {
                  "label": "Stop hook only",
                  "value": 0,
                  "n": 2
                },
                {
                  "label": "Curated + hook",
                  "value": 2,
                  "n": 2,
                  "highlight": true
                }
              ]
            },
            {
              "name": "Guessed another rate",
              "points": [
                {
                  "label": "No memory",
                  "value": 2,
                  "n": 2
                },
                {
                  "label": "/init CLAUDE.md",
                  "value": 2,
                  "n": 2
                },
                {
                  "label": "Curated, 11 lines",
                  "value": 0,
                  "n": 2
                },
                {
                  "label": "Raw notes, 60 lines",
                  "value": 0,
                  "n": 2
                },
                {
                  "label": "Dreamed notes",
                  "value": 0,
                  "n": 2
                },
                {
                  "label": "Handbook, 210 lines",
                  "value": 0,
                  "n": 2
                },
                {
                  "label": "Stop hook only",
                  "value": 2,
                  "n": 2
                },
                {
                  "label": "Curated + hook",
                  "value": 0,
                  "n": 2
                }
              ]
            }
          ],
          "note": "Late-fee task, 2 sessions per condition. \"Asked\" means the agent searched the repository, found no rate and stopped with a question instead of code.",
          "sourceIds": [
            "agent-memory-study"
          ]
        },
        {
          "id": "memory-broken-test-command",
          "title": "A stale README command: who still ran it?",
          "polarity": "lower",
          "subtitle": "Sessions that ran the README test command, which fails on Node 25 · 95% Wilson intervals",
          "kind": "dot-range",
          "unit": "rate",
          "whisker": "ci95",
          "series": [
            {
              "name": "Claude Sonnet 5.5",
              "points": [
                {
                  "label": "No memory",
                  "value": 0.8,
                  "lo": 0.5481,
                  "hi": 0.9295,
                  "n": 15
                },
                {
                  "label": "/init CLAUDE.md",
                  "value": 0.8667,
                  "lo": 0.6212,
                  "hi": 0.9626,
                  "n": 15
                },
                {
                  "label": "Curated, 11 lines",
                  "value": 0,
                  "lo": 0,
                  "hi": 0.2039,
                  "n": 15
                },
                {
                  "label": "Raw notes, 60 lines",
                  "value": 0,
                  "lo": 0,
                  "hi": 0.2039,
                  "n": 15
                },
                {
                  "label": "Dreamed notes",
                  "value": 0,
                  "lo": 0,
                  "hi": 0.2039,
                  "n": 15
                },
                {
                  "label": "Handbook, 210 lines",
                  "value": 0,
                  "lo": 0,
                  "hi": 0.2039,
                  "n": 15
                },
                {
                  "label": "Stop hook only",
                  "value": 0.6,
                  "lo": 0.3575,
                  "hi": 0.8018,
                  "n": 15
                },
                {
                  "label": "Curated + hook",
                  "value": 0,
                  "lo": 0,
                  "hi": 0.2039,
                  "n": 15
                }
              ]
            },
            {
              "name": "Claude Haiku 4.5",
              "points": [
                {
                  "label": "No memory",
                  "value": 1,
                  "lo": 0.7225,
                  "hi": 1,
                  "n": 10
                },
                {
                  "label": "/init CLAUDE.md",
                  "value": 1,
                  "lo": 0.7225,
                  "hi": 1,
                  "n": 10
                },
                {
                  "label": "Curated, 11 lines",
                  "value": 0,
                  "lo": 0,
                  "hi": 0.2775,
                  "n": 10
                },
                {
                  "label": "Raw notes, 60 lines",
                  "value": 1,
                  "lo": 0.7225,
                  "hi": 1,
                  "n": 10
                },
                {
                  "label": "Dreamed notes",
                  "value": 0.1,
                  "lo": 0.0179,
                  "hi": 0.4042,
                  "n": 10
                },
                {
                  "label": "Handbook, 210 lines",
                  "value": 0,
                  "lo": 0,
                  "hi": 0.2775,
                  "n": 10
                },
                {
                  "label": "Stop hook only",
                  "value": 1,
                  "lo": 0.7225,
                  "hi": 1,
                  "n": 10
                },
                {
                  "label": "Curated + hook",
                  "value": 0,
                  "lo": 0,
                  "hi": 0.2775,
                  "n": 10
                }
              ]
            }
          ],
          "note": "The README and npm test give a command that fails on Node 25. The curated, dreamed and handbook files name the right one. The raw notes hold both: an old \"run npm test\" note and a later correction. The /init file repeats the README.",
          "sourceIds": [
            "agent-memory-study"
          ]
        },
        {
          "id": "memory-input-tokens",
          "title": "What memory costs in context",
          "subtitle": "Median input tokens per session, cache reads included (Claude Sonnet 5.5)",
          "kind": "bar",
          "unit": "tokens",
          "series": [
            {
              "name": "Median input tokens per session",
              "points": [
                {
                  "label": "No memory",
                  "value": 78455,
                  "n": 15
                },
                {
                  "label": "/init CLAUDE.md",
                  "value": 82262,
                  "n": 15
                },
                {
                  "label": "Curated, 11 lines",
                  "value": 65045,
                  "n": 15,
                  "highlight": true
                },
                {
                  "label": "Raw notes, 60 lines",
                  "value": 76963,
                  "n": 15
                },
                {
                  "label": "Dreamed notes",
                  "value": 68218,
                  "n": 15
                },
                {
                  "label": "Handbook, 210 lines",
                  "value": 85223,
                  "n": 15
                },
                {
                  "label": "Stop hook only",
                  "value": 125674,
                  "n": 15
                },
                {
                  "label": "Curated + hook",
                  "value": 69224,
                  "n": 15
                }
              ]
            }
          ],
          "note": "Input = uncached input + cache reads + cache writes over every turn, as the CLI reports it. Most of it is read from the prompt cache. The hook adds turns: each block sends the agent back to work.",
          "sourceIds": [
            "agent-memory-study"
          ]
        },
        {
          "id": "memory-cost-per-full-pass",
          "title": "List-price cost per fully correct result (calculation)",
          "subtitle": "Sum of the CLI's cost estimates for a condition, divided by its full passes",
          "kind": "grouped-bar",
          "unit": "usd",
          "series": [
            {
              "name": "Claude Sonnet 5.5",
              "points": [
                {
                  "label": "No memory",
                  "value": 0.1386,
                  "n": 9
                },
                {
                  "label": "/init CLAUDE.md",
                  "value": 0.1359,
                  "n": 9
                },
                {
                  "label": "Curated, 11 lines",
                  "value": 0.0818,
                  "n": 15
                },
                {
                  "label": "Raw notes, 60 lines",
                  "value": 0.1009,
                  "n": 14
                },
                {
                  "label": "Dreamed notes",
                  "value": 0.0896,
                  "n": 15
                },
                {
                  "label": "Handbook, 210 lines",
                  "value": 0.1006,
                  "n": 15
                },
                {
                  "label": "Stop hook only",
                  "value": 0.1278,
                  "n": 12
                },
                {
                  "label": "Curated + hook",
                  "value": 0.0843,
                  "n": 15
                }
              ]
            },
            {
              "name": "Claude Haiku 4.5",
              "points": [
                {
                  "label": "No memory",
                  "value": 0.3786,
                  "n": 2
                },
                {
                  "label": "/init CLAUDE.md",
                  "value": 0.4317,
                  "n": 2
                },
                {
                  "label": "Curated, 11 lines",
                  "value": 0.1095,
                  "n": 7
                },
                {
                  "label": "Raw notes, 60 lines",
                  "value": 0.1255,
                  "n": 6
                },
                {
                  "label": "Dreamed notes",
                  "value": 0.1153,
                  "n": 7
                },
                {
                  "label": "Handbook, 210 lines",
                  "value": 0.2615,
                  "n": 3
                },
                {
                  "label": "Stop hook only",
                  "value": 0.1419,
                  "n": 8
                },
                {
                  "label": "Curated + hook",
                  "value": 0.096,
                  "n": 9
                }
              ]
            }
          ],
          "note": "Sessions ran on a subscription; these are the CLI's list-price estimates, not bills. A failed session still costs money, so cost per correct result falls when fewer sessions fail.",
          "sourceIds": [
            "agent-memory-study"
          ]
        },
        {
          "id": "memory-wall-time",
          "title": "Time per session",
          "subtitle": "Median wall time in seconds",
          "kind": "grouped-bar",
          "unit": "seconds",
          "series": [
            {
              "name": "Claude Sonnet 5.5",
              "points": [
                {
                  "label": "No memory",
                  "value": 18,
                  "n": 15
                },
                {
                  "label": "/init CLAUDE.md",
                  "value": 19,
                  "n": 15
                },
                {
                  "label": "Curated, 11 lines",
                  "value": 21.9,
                  "n": 15
                },
                {
                  "label": "Raw notes, 60 lines",
                  "value": 26.8,
                  "n": 15
                },
                {
                  "label": "Dreamed notes",
                  "value": 27.2,
                  "n": 15
                },
                {
                  "label": "Handbook, 210 lines",
                  "value": 23.6,
                  "n": 15
                },
                {
                  "label": "Stop hook only",
                  "value": 27.3,
                  "n": 15
                },
                {
                  "label": "Curated + hook",
                  "value": 22,
                  "n": 15
                }
              ]
            },
            {
              "name": "Claude Haiku 4.5",
              "points": [
                {
                  "label": "No memory",
                  "value": 54.2,
                  "n": 10
                },
                {
                  "label": "/init CLAUDE.md",
                  "value": 52.9,
                  "n": 10
                },
                {
                  "label": "Curated, 11 lines",
                  "value": 51.7,
                  "n": 10
                },
                {
                  "label": "Raw notes, 60 lines",
                  "value": 51,
                  "n": 10
                },
                {
                  "label": "Dreamed notes",
                  "value": 51.8,
                  "n": 10
                },
                {
                  "label": "Handbook, 210 lines",
                  "value": 49.9,
                  "n": 10
                },
                {
                  "label": "Stop hook only",
                  "value": 68.5,
                  "n": 10
                },
                {
                  "label": "Curated + hook",
                  "value": 52.8,
                  "n": 10
                }
              ]
            }
          ],
          "note": "Up to four sessions ran at a time on one machine. Sessions without memory were often shorter because they stopped to ask or skipped the changelog.",
          "sourceIds": [
            "agent-memory-study"
          ]
        },
        {
          "id": "memory-dreaming-scorecard",
          "title": "Dreaming: what one consolidation pass kept and dropped",
          "subtitle": "Fixed-pattern checks on each memory file: 10 current facts, 5 stale notes, 9 transient notes",
          "kind": "grouped-bar",
          "unit": "count",
          "series": [
            {
              "name": "Current facts kept (of 10)",
              "points": [
                {
                  "label": "Raw notes",
                  "value": 10
                },
                {
                  "label": "Dream 1",
                  "value": 10
                },
                {
                  "label": "Dream 2",
                  "value": 10
                },
                {
                  "label": "Dream 3",
                  "value": 10
                },
                {
                  "label": "Curated",
                  "value": 10
                },
                {
                  "label": "/init",
                  "value": 5
                }
              ]
            },
            {
              "name": "Stale notes left (of 5)",
              "points": [
                {
                  "label": "Raw notes",
                  "value": 5
                },
                {
                  "label": "Dream 1",
                  "value": 0
                },
                {
                  "label": "Dream 2",
                  "value": 0
                },
                {
                  "label": "Dream 3",
                  "value": 0
                },
                {
                  "label": "Curated",
                  "value": 0
                },
                {
                  "label": "/init",
                  "value": 1
                }
              ]
            },
            {
              "name": "Transient notes left (of 9)",
              "points": [
                {
                  "label": "Raw notes",
                  "value": 9
                },
                {
                  "label": "Dream 1",
                  "value": 1
                },
                {
                  "label": "Dream 2",
                  "value": 1
                },
                {
                  "label": "Dream 3",
                  "value": 1
                },
                {
                  "label": "Curated",
                  "value": 0
                },
                {
                  "label": "/init",
                  "value": 0
                }
              ]
            }
          ],
          "note": "Each dream is one Claude Sonnet call with no tools over the 56 raw notes. A stale note that the new file marks as replaced does not count as left. The one transient note every dream kept is the current dev-server port.",
          "sourceIds": [
            "agent-memory-study"
          ]
        }
      ],
      "tables": [
        {
          "id": "memory-by-condition",
          "title": "Every condition, Claude Sonnet 5.5",
          "columns": [
            {
              "key": "condition",
              "label": "Memory",
              "unit": "text"
            },
            {
              "key": "full",
              "label": "Full pass",
              "unit": "text"
            },
            {
              "key": "functional",
              "label": "Hidden tests pass",
              "unit": "text"
            },
            {
              "key": "code",
              "label": "Code-visible rules",
              "unit": "text"
            },
            {
              "key": "structure",
              "label": "Folder-visible rules",
              "unit": "text"
            },
            {
              "key": "tribal",
              "label": "Team knowledge",
              "unit": "text"
            },
            {
              "key": "broken",
              "label": "Ran the stale test command",
              "unit": "text"
            },
            {
              "key": "input",
              "label": "Median input tokens",
              "unit": "tokens"
            },
            {
              "key": "wall",
              "label": "Median time (s)",
              "unit": "seconds"
            },
            {
              "key": "hook",
              "label": "Hook blocks",
              "unit": "count"
            }
          ],
          "rows": [
            {
              "condition": "No memory",
              "full": "9/15",
              "functional": "12/15",
              "code": "57/60",
              "structure": "21/21",
              "tribal": "6/15",
              "broken": "12/15",
              "input": 78455,
              "wall": 18,
              "hook": 0
            },
            {
              "condition": "/init CLAUDE.md",
              "full": "9/15",
              "functional": "12/15",
              "code": "57/60",
              "structure": "21/21",
              "tribal": "10/15",
              "broken": "13/15",
              "input": 82262,
              "wall": 19,
              "hook": 0
            },
            {
              "condition": "Curated, 11 lines",
              "full": "15/15",
              "functional": "15/15",
              "code": "60/60",
              "structure": "21/21",
              "tribal": "15/15",
              "broken": "0/15",
              "input": 65045,
              "wall": 21.9,
              "hook": 0
            },
            {
              "condition": "Raw notes, 60 lines",
              "full": "14/15",
              "functional": "14/15",
              "code": "60/60",
              "structure": "21/21",
              "tribal": "15/15",
              "broken": "0/15",
              "input": 76963,
              "wall": 26.8,
              "hook": 0
            },
            {
              "condition": "Dreamed notes",
              "full": "15/15",
              "functional": "15/15",
              "code": "60/60",
              "structure": "21/21",
              "tribal": "15/15",
              "broken": "0/15",
              "input": 68218,
              "wall": 27.2,
              "hook": 0
            },
            {
              "condition": "Handbook, 210 lines",
              "full": "15/15",
              "functional": "15/15",
              "code": "60/60",
              "structure": "21/21",
              "tribal": "15/15",
              "broken": "0/15",
              "input": 85223,
              "wall": 23.6,
              "hook": 0
            },
            {
              "condition": "Stop hook only",
              "full": "12/15",
              "functional": "12/15",
              "code": "60/60",
              "structure": "21/21",
              "tribal": "10/15",
              "broken": "9/15",
              "input": 125674,
              "wall": 27.3,
              "hook": 5
            },
            {
              "condition": "Curated + hook",
              "full": "15/15",
              "functional": "15/15",
              "code": "60/60",
              "structure": "21/21",
              "tribal": "15/15",
              "broken": "0/15",
              "input": 69224,
              "wall": 22,
              "hook": 0
            }
          ]
        },
        {
          "id": "memory-by-condition-haiku",
          "title": "Every condition, Claude Haiku 4.5",
          "columns": [
            {
              "key": "condition",
              "label": "Memory",
              "unit": "text"
            },
            {
              "key": "full",
              "label": "Full pass",
              "unit": "text"
            },
            {
              "key": "functional",
              "label": "Hidden tests pass",
              "unit": "text"
            },
            {
              "key": "code",
              "label": "Code-visible rules",
              "unit": "text"
            },
            {
              "key": "structure",
              "label": "Folder-visible rules",
              "unit": "text"
            },
            {
              "key": "tribal",
              "label": "Team knowledge",
              "unit": "text"
            },
            {
              "key": "broken",
              "label": "Ran the stale test command",
              "unit": "text"
            },
            {
              "key": "input",
              "label": "Median input tokens",
              "unit": "tokens"
            },
            {
              "key": "wall",
              "label": "Median time (s)",
              "unit": "seconds"
            },
            {
              "key": "hook",
              "label": "Hook blocks",
              "unit": "count"
            }
          ],
          "rows": [
            {
              "condition": "No memory",
              "full": "2/10",
              "functional": "6/10",
              "code": "40/40",
              "structure": "12/14",
              "tribal": "0/10",
              "broken": "10/10",
              "input": 301233,
              "wall": 54.2,
              "hook": 0
            },
            {
              "condition": "/init CLAUDE.md",
              "full": "2/10",
              "functional": "7/10",
              "code": "38/40",
              "structure": "14/14",
              "tribal": "1/10",
              "broken": "10/10",
              "input": 323519,
              "wall": 52.9,
              "hook": 0
            },
            {
              "condition": "Curated, 11 lines",
              "full": "7/10",
              "functional": "9/10",
              "code": "40/40",
              "structure": "14/14",
              "tribal": "8/10",
              "broken": "0/10",
              "input": 284212,
              "wall": 51.7,
              "hook": 0
            },
            {
              "condition": "Raw notes, 60 lines",
              "full": "6/10",
              "functional": "10/10",
              "code": "40/40",
              "structure": "14/14",
              "tribal": "6/10",
              "broken": "10/10",
              "input": 285263,
              "wall": 51,
              "hook": 0
            },
            {
              "condition": "Dreamed notes",
              "full": "7/10",
              "functional": "9/10",
              "code": "40/40",
              "structure": "14/14",
              "tribal": "8/10",
              "broken": "1/10",
              "input": 267405,
              "wall": 51.8,
              "hook": 0
            },
            {
              "condition": "Handbook, 210 lines",
              "full": "3/10",
              "functional": "10/10",
              "code": "40/40",
              "structure": "14/14",
              "tribal": "3/10",
              "broken": "0/10",
              "input": 313796,
              "wall": 49.9,
              "hook": 0
            },
            {
              "condition": "Stop hook only",
              "full": "8/10",
              "functional": "8/10",
              "code": "40/40",
              "structure": "14/14",
              "tribal": "8/10",
              "broken": "10/10",
              "input": 530935,
              "wall": 68.5,
              "hook": 10
            },
            {
              "condition": "Curated + hook",
              "full": "9/10",
              "functional": "9/10",
              "code": "40/40",
              "structure": "14/14",
              "tribal": "10/10",
              "broken": "0/10",
              "input": 340491,
              "wall": 52.8,
              "hook": 3
            }
          ]
        },
        {
          "id": "memory-raw-notes",
          "title": "The 56 raw notes, labelled by what a consolidation pass should do with each (paths in the test repository shown under tally/)",
          "columns": [
            {
              "key": "date",
              "label": "Saved",
              "unit": "text"
            },
            {
              "key": "note",
              "label": "Note",
              "unit": "text"
            },
            {
              "key": "fate",
              "label": "Should be",
              "unit": "text"
            }
          ],
          "rows": [
            {
              "date": "2026-06-12",
              "note": "Project is `tally`, a small invoicing ledger for the billing service. Plain ESM, no dependencies.",
              "fate": "kept"
            },
            {
              "date": "2026-06-12",
              "note": "Shah prefers small PRs with one change each.",
              "fate": "kept"
            },
            {
              "date": "2026-06-14",
              "note": "Dev server for the billing service ran on port 3001 this afternoon.",
              "fate": "removed: transient"
            },
            {
              "date": "2026-06-18",
              "note": "Run tests with `npm test`.",
              "fate": "removed: stale"
            },
            {
              "date": "2026-06-20",
              "note": "Amounts are dollars as floats. Format them with `toFixed(2)`.",
              "fate": "removed: stale"
            },
            {
              "date": "2026-06-20",
              "note": "Use 2-space indentation and single quotes.",
              "fate": "kept"
            },
            {
              "date": "2026-06-22",
              "note": "Branch `feat/customers-v1` pushed; waiting for review.",
              "fate": "removed: transient"
            },
            {
              "date": "2026-06-25",
              "note": "The time helper is `now()` in `tally/src/clock.js`.",
              "fate": "removed: stale"
            },
            {
              "date": "2026-06-27",
              "note": "Shah is out on Friday; do not merge anything then.",
              "fate": "removed: transient"
            },
            {
              "date": "2026-07-01",
              "note": "Customers live in `tally/src/customers.js`.",
              "fate": "kept"
            },
            {
              "date": "2026-07-03",
              "note": "Throw `TallyError` with a code from `CODES` in `tally/src/errors.js`; the billing service maps codes to HTTP statuses.",
              "fate": "kept"
            },
            {
              "date": "2026-07-03",
              "note": "Error codes are public API. Do not rename an existing code.",
              "fate": "kept"
            },
            {
              "date": "2026-07-05",
              "note": "Prefer named exports.",
              "fate": "kept"
            },
            {
              "date": "2026-07-08",
              "note": "Used jq to inspect the payment fixtures; worked fine.",
              "fate": "removed: transient"
            },
            {
              "date": "2026-07-10",
              "note": "Do not use `console.log` in `tally/src/`; use `log()` from `tally/src/log.js`.",
              "fate": "kept"
            },
            {
              "date": "2026-07-12",
              "note": "Staging database was reset today.",
              "fate": "removed: transient"
            },
            {
              "date": "2026-07-14",
              "note": "Late fee is 2% of the unpaid balance.",
              "fate": "removed: stale"
            },
            {
              "date": "2026-07-15",
              "note": "PR #212 (partial payments) merged.",
              "fate": "removed: transient"
            },
            {
              "date": "2026-07-18",
              "note": "The report module is new; Shah wants it in the next release.",
              "fate": "kept"
            },
            {
              "date": "2026-07-22",
              "note": "Shah wants a line under `## Unreleased` in CHANGELOG.md for every user-visible change.",
              "fate": "kept"
            },
            {
              "date": "2026-07-22",
              "note": "Keep CHANGELOG entries to one line.",
              "fate": "kept"
            },
            {
              "date": "2026-07-25",
              "note": "`slugify()` is used for export file names.",
              "fate": "kept"
            },
            {
              "date": "2026-07-29",
              "note": "Remember: no `console.log` in src, use the log helper.",
              "fate": "merged: repeat"
            },
            {
              "date": "2026-08-01",
              "note": "Released 1.3.0 (monthly report, partial payments).",
              "fate": "removed: transient"
            },
            {
              "date": "2026-08-03",
              "note": "CI was slow today because of a runner outage.",
              "fate": "removed: transient"
            },
            {
              "date": "2026-08-05",
              "note": "Incident: someone edited migration 0002 and production schemas drifted. Migrations are append-only now: add a new `tally/migrations/NNNN_<name>.json`, never edit an old one.",
              "fate": "kept"
            },
            {
              "date": "2026-08-07",
              "note": "Store loads the schema from `tally/migrations/*.json` in file-name order.",
              "fate": "kept"
            },
            {
              "date": "2026-08-09",
              "note": "Shah prefers early returns over nested ifs.",
              "fate": "kept"
            },
            {
              "date": "2026-08-12",
              "note": "Migrated all money to integer cents (`amountCents`). Parse input with `parseAmount()`, print with `formatCents()`. No floats for money.",
              "fate": "kept"
            },
            {
              "date": "2026-08-12",
              "note": "`toFixed` caused rounding bugs in the old float code.",
              "fate": "kept"
            },
            {
              "date": "2026-08-14",
              "note": "Reminder: throw TallyError, not Error.",
              "fate": "merged: repeat"
            },
            {
              "date": "2026-08-16",
              "note": "Coffee machine on floor 3 is broken (Shah mentioned it).",
              "fate": "removed: transient"
            },
            {
              "date": "2026-08-19",
              "note": "CI flaky again; reran twice.",
              "fate": "removed: transient"
            },
            {
              "date": "2026-08-22",
              "note": "Payment amounts must not exceed the open balance (`E_OVERPAYMENT`).",
              "fate": "kept"
            },
            {
              "date": "2026-08-24",
              "note": "Branch `fix/report-totals` pushed.",
              "fate": "removed: transient"
            },
            {
              "date": "2026-08-27",
              "note": "Integer cents everywhere. Do not divide by 100 except when formatting.",
              "fate": "merged: repeat"
            },
            {
              "date": "2026-08-30",
              "note": "Renamed the time helper: use `nowIso()` from `tally/src/time.js`. Tests inject a clock with `setClock()`. Do not call `new Date()` or `Date.now()` in src.",
              "fate": "kept"
            },
            {
              "date": "2026-09-02",
              "note": "Throw TallyError with a CODES entry; add a new code when none fits.",
              "fate": "kept"
            },
            {
              "date": "2026-09-04",
              "note": "Shah likes tests next to each feature change.",
              "fate": "kept"
            },
            {
              "date": "2026-09-06",
              "note": "Looked at Stripe's refund API for ideas; not needed yet.",
              "fate": "removed: transient"
            },
            {
              "date": "2026-09-08",
              "note": "Port 3001 is taken by another service now; billing dev server moved to 3002.",
              "fate": "removed: transient"
            },
            {
              "date": "2026-09-10",
              "note": "Lost a fix because I edited `tally/src/report.gen.js` directly; it is generated. Edit `tally/templates/report.tmpl.js` and run `node tally/scripts/gen-report.mjs`.",
              "fate": "kept"
            },
            {
              "date": "2026-09-11",
              "note": "The report template uses a `__CURRENCY__` placeholder.",
              "fate": "kept"
            },
            {
              "date": "2026-09-13",
              "note": "Shah asked for clearer error messages in validation.",
              "fate": "kept"
            },
            {
              "date": "2026-09-15",
              "note": "Finance changed the late fee to 1.25% of the unpaid balance, effective now. Round half up to the cent.",
              "fate": "kept"
            },
            {
              "date": "2026-09-16",
              "note": "Email validation added to customers.",
              "fate": "kept"
            },
            {
              "date": "2026-09-18",
              "note": "Reminder: changelog line for every user-visible change.",
              "fate": "merged: repeat"
            },
            {
              "date": "2026-09-20",
              "note": "Released 1.4.0 (integer cents, email validation).",
              "fate": "removed: transient"
            },
            {
              "date": "2026-09-21",
              "note": "`node --test test/` fails on Node 25 (it treats the folder as a file). Run `node --test` with no path. `npm test` uses the broken form.",
              "fate": "kept"
            },
            {
              "date": "2026-09-23",
              "note": "Prefer `Object.freeze` for constant maps.",
              "fate": "kept"
            },
            {
              "date": "2026-09-24",
              "note": "Do not edit migration files that already exist.",
              "fate": "merged: repeat"
            },
            {
              "date": "2026-09-26",
              "note": "Shah reviews PRs in the morning.",
              "fate": "kept"
            },
            {
              "date": "2026-09-28",
              "note": "Use the log helper for audit events (payments, refunds).",
              "fate": "kept"
            },
            {
              "date": "2026-09-30",
              "note": "The billing service reads `CODES` at start-up; new codes need no other change.",
              "fate": "kept"
            },
            {
              "date": "2026-10-01",
              "note": "Staging is on the new cluster.",
              "fate": "removed: transient"
            },
            {
              "date": "2026-10-02",
              "note": "Generated file again: never hand-edit `tally/src/report.gen.js`.",
              "fate": "merged: repeat"
            }
          ]
        },
        {
          "id": "memory-by-task",
          "title": "Full passes per task (Claude Sonnet 5.5, of 3)",
          "columns": [
            {
              "key": "task",
              "label": "Task",
              "unit": "text"
            },
            {
              "key": "none",
              "label": "No memory",
              "unit": "count"
            },
            {
              "key": "init",
              "label": "/init CLAUDE.md",
              "unit": "count"
            },
            {
              "key": "curated",
              "label": "Curated, 11 lines",
              "unit": "count"
            },
            {
              "key": "raw",
              "label": "Raw notes, 60 lines",
              "unit": "count"
            },
            {
              "key": "dreamed",
              "label": "Dreamed notes",
              "unit": "count"
            },
            {
              "key": "bloated",
              "label": "Handbook, 210 lines",
              "unit": "count"
            },
            {
              "key": "hook",
              "label": "Stop hook only",
              "unit": "count"
            },
            {
              "key": "curated-hook",
              "label": "Curated + hook",
              "unit": "count"
            }
          ],
          "rows": [
            {
              "task": "Refunds",
              "none": 3,
              "init": 3,
              "curated": 3,
              "raw": 2,
              "dreamed": 3,
              "bloated": 3,
              "hook": 3,
              "curated-hook": 3
            },
            {
              "task": "Report decimals",
              "none": 0,
              "init": 0,
              "curated": 3,
              "raw": 3,
              "dreamed": 3,
              "bloated": 3,
              "hook": 3,
              "curated-hook": 3
            },
            {
              "task": "Customer phone",
              "none": 3,
              "init": 3,
              "curated": 3,
              "raw": 3,
              "dreamed": 3,
              "bloated": 3,
              "hook": 3,
              "curated-hook": 3
            },
            {
              "task": "Late fee",
              "none": 0,
              "init": 0,
              "curated": 3,
              "raw": 3,
              "dreamed": 3,
              "bloated": 3,
              "hook": 0,
              "curated-hook": 3
            },
            {
              "task": "Slugify (control)",
              "none": 3,
              "init": 3,
              "curated": 3,
              "raw": 3,
              "dreamed": 3,
              "bloated": 3,
              "hook": 3,
              "curated-hook": 3
            }
          ]
        }
      ],
      "related": [
        "caching-consistency",
        "cli-model-latency-tokens"
      ],
      "hero": {
        "statIds": [
          "memory-team-knowledge-none",
          "memory-team-knowledge-curated"
        ]
      }
    },
    {
      "slug": "system-one-arena",
      "title": "System One arena: Jev vs Clef and five open decision models, head to head",
      "seoTitle": "Jev vs Clef: 7 decision models on 1,085 decisions",
      "description": "Jev 1.13, Clef 27B, Clef-Flash and four small open decision models on 1,085 checkable decisions and in head-to-head games, Pong included.",
      "question": "How good are typed-decision models at decisions with a checkable answer, and which one wins when they play each other?",
      "answer": "Jev 1.13 answered 76.8% of 1,048 graded decisions correctly, clearly ahead of every other model (exact McNemar p < 0.001 for each pair). The best open model, Clef 27B, reached 69.9%. At one provider's list prices (Cloudflare Workers AI) a thousand decisions cost $0.035 for Jev 1.13, $0.051 for Clef-Flash 9B, $0.136 for Clef 27B. Speed is compared only where the footing is the same: on one GPU a third-party board measured Clef 27B at 209 ms and Clef-Flash 9B at 39 ms, while Jev 1.13, a hosted API, took 524 ms in that harness with the network included. Clef-Flash 9B (6.49 GB, 65.2%) and Kev 4B (3.03 GB, 64.0%) were statistically tied. Laya (0.45 GB, 27.4%) and Julia-1 (0.17 GB, 25.9%) were statistically tied. When no option fitted, Jev 1.13 chose \"none\", \"ask\" or \"escalate\" 86% of the time; Laya 18%. Clef 27B and Clef-Flash 9B never changed a decision when the options were shuffled, but changed 18% and 18% when the same options were renamed a, b, c. In 1,512 round-robin games (tic-tac-toe, Connect Four, Nim, Dots and Boxes and Pong), Jev 1.13 rated highest (Elo 1040 against 896 for the random player; the other models sit between 877 and 941). Jev 1.13 (30–11 of 42) and Clef-Flash 9B (29–11 of 42) beat the random player clearly (exact sign test p < 0.05); the other models did not. In Pong, where an answer counts only once it arrives, Jev 1.13 won 13 of 16 games and Kev 4B 12, while Clef 27B, the most accurate open model on the exam, won 4 at 1.8 s per decision. In the arena leagues Jev 1.13 ranked as follows: Othello: rank 9 of 9, Elo 890; best was perfect at 1280; the random player 1015; Pong (every answer after 150 ms): rank 3 of 8, Elo 1042; best was perfect at 1098; the random player 930; Snake (every answer after 150 ms): rank 2 of 9, Elo 1103; best was expert at 1185; the random player 954; Tron (every answer after 150 ms): rank 2 of 9, Elo 1219; best was expert at 1267; the random player 898.",
      "date": "2026-10-06",
      "updated": "2026-10-06",
      "tags": [
        "jev",
        "clef",
        "decision-models",
        "system-one",
        "llama-cpp",
        "open-models",
        "routing"
      ],
      "method": [
        "Seven models with the same API (a JSON state in, a choice or a yes/no with probabilities out): TypeSafe Jev 1.13 over its hosted API, and Cloudflare Clef 27B, Clef-Flash 9B, lev 4B, Kev 4B, Laya and Julia-1 as GGUF files (Q4_K_M or Q8_0, checked against the published SHA-256) in llama.cpp 0.6.0 on one Mac Studio (M3 Ultra, 96 GB).",
        "Five suites, 1,085 items, written by five agents that never called a model: games solved by search (tic-tac-toe, Connect Four, Nim, Pong intercepts, Sudoku, Wordle, mazes, Hanoi, coin weighing, knight moves), logic and thought experiments (syllogisms, knights and knaves, Monty Hall, base rates, expected value, bias probes, framing pairs), fictional company policies in 20 industries, usability intents on 14 kinds of app, and stress tests (prompt injection, long logs, thresholds, near-identical options, paraphrases).",
        "Every answer comes from search, arithmetic, logic or a rule written in the item. Each suite has its own verifier that re-derives every answer. A second agent re-solved every item of the first versions of the suites (978) and found no wrong answer; the ambiguities and shortcuts it found were fixed before the freeze, adding 107 items and rewriting others. Consistency-only items (no right answer) are scored only on agreement inside their group.",
        "Exam pass: every item, every model; choice items also with shuffled options and with options renamed a, b, c. Latency pass: 120 seeded items, one model and one call at a time. Repeat pass: 60 items asked twice.",
        "Matches: a round robin of tic-tac-toe, Connect Four, Nim and Dots and Boxes (10 games per pairing, colours swapped), real-time Pong (first to 3 points after amendment 2) where each answer takes effect only after its measured latency, and a random and a perfect player as anchors. Elo: K 32, averaged over 200 seeded game orders.",
        "The protocol, the item hashes and the game counts were declared before the first counted call. Two amendments, both declared before the data they affect, are in the published protocol: the local-server batch size (amendment 1) and the Pong match length (amendment 2). One erratum: the Connect Four reference solver graded some moves in a search-order-dependent way; it was fixed and every move regraded, and no pick, probability, result or rating changed (erratum 2). Every call and every game is kept; errors count as wrong."
      ],
      "caveats": [
        "Local models ran quantized on one Mac; the vendors measured full-precision weights on data-centre GPUs. Latency on other hardware will differ, and Jev latency includes the network.",
        "The items are synthetic and written by agents. They test clear, checkable decisions, not the messy decisions a product meets; real accuracy depends on how well the state and options are written.",
        "Typed-decision models are built for routing and gating with a confidence threshold. These items have no threshold: a model that is unsure must still pick.",
        "A few small families lean on content where the right option is often the longest or shortest description (sudoku, seating order, gambles). Reported, not removed.",
        "These are llama.cpp ports, not the reference implementations: llama.cpp converts only part of lev's output head, an open llama.cpp issue reports degraded probabilities for Clef Q8_0 (we ran Q4_K_M), and Laya and Julia-1 were trained on inputs of 512 to 1,024 tokens while our items average about 800.",
        "One harness error was fixed mid-run (amendment 1): the local servers first ran with a 512-token batch, which rejected longer inputs. Every local model was rerun from the start with a whole-input batch, one model at a time; the earlier rows are kept apart and not counted.",
        "Games: 10 games per pairing in each turn-based game (colours swapped), 2 in Pong, first to 3 after amendment 2. The sign tests against the random player are not corrected for the many comparisons made, so read them as descriptions. Connect Four \"perfect\" is a depth-10 search plus exact endgames, not a full solve.",
        "In Pong each answer moves the paddle only after its measured latency, and the arrival point of the ball is in the state, so Pong tests reading and speed more than physics. The viral \"smarter AI lost at Pong\" result (Laya beating Jev) did not reproduce under these rules: Jev won all 4 games of the featured series.",
        "Jev ran on TypeSafe's servers and the six open models on one Mac Studio, so speed across that line is not comparable. This study compares capability (same questions and positions) directly, and speed only on one machine, on one GPU (third-party numbers), through one gateway (OpenRouter's numbers) or as list price.",
        "Arena leagues (Othello, and Tron, Snake and Pong in fair mode) are content recordings, not tests of the models: one seed, 2 to 4 games per pairing, protocol amendment 3. The Mac also ran other work while they were recorded (erratum 4), so their call times are not clean measurements and no speed claim uses them; fair-mode results do not depend on call time."
      ],
      "sourceIds": [
        "system-one-arena"
      ],
      "stats": [
        {
          "id": "arena-items",
          "label": "Decision items, written by agents that never called a model (1,048 graded, 37 consistency-only)",
          "value": 1085,
          "unit": "count",
          "display": "1,085",
          "n": 1085
        },
        {
          "id": "arena-calls",
          "label": "Decision calls made and counted (exam, latency and repeat passes)",
          "value": 23247,
          "unit": "calls",
          "display": "23,247",
          "n": 23247
        },
        {
          "id": "arena-best",
          "label": "Highest accuracy on the graded items: Jev 1.13",
          "value": 0.7681,
          "unit": "rate",
          "display": "76.8% (805/1048)",
          "n": 1048,
          "ci": [
            0.7416,
            0.7927
          ]
        },
        {
          "id": "arena-best-open",
          "label": "Best open model: Clef 27B",
          "value": 0.6994,
          "unit": "rate",
          "display": "69.9% (733/1048)",
          "n": 1048,
          "ci": [
            0.671,
            0.7264
          ]
        },
        {
          "id": "arena-smallest",
          "label": "Smallest model: Julia-1 (0.17 GB file)",
          "value": 0.2586,
          "unit": "rate",
          "display": "25.9% (271/1048)",
          "n": 1048,
          "ci": [
            0.233,
            0.2859
          ]
        },
        {
          "id": "arena-jev-cost",
          "label": "Jev 1.13 list-price cost per 1,000 decisions (calculation from its reported input tokens)",
          "value": 0.03465,
          "unit": "usd",
          "display": "$0.0347",
          "n": 3081
        },
        {
          "id": "arena-jev-repeat",
          "label": "Same question asked twice: Jev 1.13 gave the same answer 58/60 times; the local models every time",
          "value": 0.9667,
          "unit": "rate",
          "display": "58/60",
          "n": 60
        },
        {
          "id": "arena-jev-latency",
          "label": "Jev 1.13 hosted API: median call time from Houston, network included (not comparable with a model on another machine)",
          "value": 137,
          "unit": "ms",
          "display": "137 ms",
          "n": 120
        },
        {
          "id": "arena-chance",
          "label": "Expected accuracy of picking an option at random on the same graded items (calculation)",
          "value": 0.224,
          "unit": "rate",
          "display": "22.4%",
          "n": 1048
        },
        {
          "id": "arena-rank-agreement",
          "label": "Rank agreement between our accuracy order and the independent Decision Index 0.2.1 order, same seven models (Spearman, calculation)",
          "value": 0.93,
          "unit": "score",
          "display": "0.93",
          "n": 7
        },
        {
          "id": "arena-games",
          "label": "Games played to the end and counted (round robin, featured series and speed ladder)",
          "value": 1620,
          "unit": "count",
          "display": "1,620",
          "n": 1620
        }
      ],
      "charts": [
        {
          "id": "arena-accuracy",
          "title": "Who decides right? Accuracy on 1,000+ checkable decisions",
          "subtitle": "Share of graded items answered correctly, first presentation · 95% Wilson intervals",
          "kind": "dot-range",
          "unit": "rate",
          "whisker": "ci95",
          "series": [
            {
              "name": "Accuracy",
              "points": [
                {
                  "label": "Jev 1.13",
                  "value": 0.7681,
                  "lo": 0.7416,
                  "hi": 0.7927,
                  "n": 1048,
                  "highlight": true
                },
                {
                  "label": "Clef 27B",
                  "value": 0.6994,
                  "lo": 0.671,
                  "hi": 0.7264,
                  "n": 1048
                },
                {
                  "label": "Clef-Flash 9B",
                  "value": 0.6517,
                  "lo": 0.6224,
                  "hi": 0.68,
                  "n": 1048
                },
                {
                  "label": "Kev 4B",
                  "value": 0.6403,
                  "lo": 0.6107,
                  "hi": 0.6688,
                  "n": 1048
                },
                {
                  "label": "lev 4B",
                  "value": 0.5859,
                  "lo": 0.5558,
                  "hi": 0.6153,
                  "n": 1048
                },
                {
                  "label": "Laya",
                  "value": 0.2739,
                  "lo": 0.2477,
                  "hi": 0.3016,
                  "n": 1048
                },
                {
                  "label": "Julia-1",
                  "value": 0.2586,
                  "lo": 0.233,
                  "hi": 0.2859,
                  "n": 1048
                }
              ]
            }
          ],
          "note": "1048 graded items in five suites. An error, a timeout or a label outside the option set counts as wrong. Two models differ clearly only where the paired McNemar test says so (table below).",
          "sourceIds": [
            "system-one-arena"
          ]
        },
        {
          "id": "arena-by-suite",
          "title": "Where each model is strong",
          "subtitle": "Accuracy by suite, first presentation",
          "kind": "grouped-bar",
          "unit": "rate",
          "series": [
            {
              "name": "Jev 1.13",
              "points": [
                {
                  "label": "Games",
                  "value": 0.4219,
                  "lo": 0.3542,
                  "hi": 0.4926,
                  "n": 192
                },
                {
                  "label": "Logic and thought experiments",
                  "value": 0.7436,
                  "lo": 0.6698,
                  "hi": 0.8057,
                  "n": 156
                },
                {
                  "label": "Policy cases, 20 industries",
                  "value": 0.8139,
                  "lo": 0.7636,
                  "hi": 0.8555,
                  "n": 274
                },
                {
                  "label": "Usability intents",
                  "value": 0.9336,
                  "lo": 0.8934,
                  "hi": 0.9594,
                  "n": 226
                },
                {
                  "label": "Stress tests",
                  "value": 0.87,
                  "lo": 0.8163,
                  "hi": 0.9097,
                  "n": 200
                }
              ]
            },
            {
              "name": "Clef 27B",
              "points": [
                {
                  "label": "Games",
                  "value": 0.3594,
                  "lo": 0.2949,
                  "hi": 0.4294,
                  "n": 192
                },
                {
                  "label": "Logic and thought experiments",
                  "value": 0.6474,
                  "lo": 0.5697,
                  "hi": 0.718,
                  "n": 156
                },
                {
                  "label": "Policy cases, 20 industries",
                  "value": 0.719,
                  "lo": 0.663,
                  "hi": 0.7689,
                  "n": 274
                },
                {
                  "label": "Usability intents",
                  "value": 0.8894,
                  "lo": 0.8418,
                  "hi": 0.9239,
                  "n": 226
                },
                {
                  "label": "Stress tests",
                  "value": 0.825,
                  "lo": 0.7664,
                  "hi": 0.8714,
                  "n": 200
                }
              ]
            },
            {
              "name": "Clef-Flash 9B",
              "points": [
                {
                  "label": "Games",
                  "value": 0.3073,
                  "lo": 0.2463,
                  "hi": 0.3758,
                  "n": 192
                },
                {
                  "label": "Logic and thought experiments",
                  "value": 0.5833,
                  "lo": 0.5049,
                  "hi": 0.6578,
                  "n": 156
                },
                {
                  "label": "Policy cases, 20 industries",
                  "value": 0.7226,
                  "lo": 0.6668,
                  "hi": 0.7723,
                  "n": 274
                },
                {
                  "label": "Usability intents",
                  "value": 0.8673,
                  "lo": 0.8168,
                  "hi": 0.9054,
                  "n": 226
                },
                {
                  "label": "Stress tests",
                  "value": 0.695,
                  "lo": 0.628,
                  "hi": 0.7546,
                  "n": 200
                }
              ]
            },
            {
              "name": "Kev 4B",
              "points": [
                {
                  "label": "Games",
                  "value": 0.3854,
                  "lo": 0.3195,
                  "hi": 0.4559,
                  "n": 192
                },
                {
                  "label": "Logic and thought experiments",
                  "value": 0.5449,
                  "lo": 0.4666,
                  "hi": 0.621,
                  "n": 156
                },
                {
                  "label": "Policy cases, 20 industries",
                  "value": 0.6752,
                  "lo": 0.6176,
                  "hi": 0.7279,
                  "n": 274
                },
                {
                  "label": "Usability intents",
                  "value": 0.8009,
                  "lo": 0.744,
                  "hi": 0.8477,
                  "n": 226
                },
                {
                  "label": "Stress tests",
                  "value": 0.73,
                  "lo": 0.6646,
                  "hi": 0.7868,
                  "n": 200
                }
              ]
            },
            {
              "name": "lev 4B",
              "points": [
                {
                  "label": "Games",
                  "value": 0.3021,
                  "lo": 0.2415,
                  "hi": 0.3704,
                  "n": 192
                },
                {
                  "label": "Logic and thought experiments",
                  "value": 0.5064,
                  "lo": 0.4287,
                  "hi": 0.5838,
                  "n": 156
                },
                {
                  "label": "Policy cases, 20 industries",
                  "value": 0.5912,
                  "lo": 0.5322,
                  "hi": 0.6478,
                  "n": 274
                },
                {
                  "label": "Usability intents",
                  "value": 0.823,
                  "lo": 0.768,
                  "hi": 0.8672,
                  "n": 226
                },
                {
                  "label": "Stress tests",
                  "value": 0.645,
                  "lo": 0.5765,
                  "hi": 0.708,
                  "n": 200
                }
              ]
            },
            {
              "name": "Laya",
              "points": [
                {
                  "label": "Games",
                  "value": 0.1771,
                  "lo": 0.1296,
                  "hi": 0.2373,
                  "n": 192
                },
                {
                  "label": "Logic and thought experiments",
                  "value": 0.2821,
                  "lo": 0.2173,
                  "hi": 0.3572,
                  "n": 156
                },
                {
                  "label": "Policy cases, 20 industries",
                  "value": 0.2664,
                  "lo": 0.2176,
                  "hi": 0.3217,
                  "n": 274
                },
                {
                  "label": "Usability intents",
                  "value": 0.323,
                  "lo": 0.2654,
                  "hi": 0.3865,
                  "n": 226
                },
                {
                  "label": "Stress tests",
                  "value": 0.315,
                  "lo": 0.2546,
                  "hi": 0.3823,
                  "n": 200
                }
              ]
            },
            {
              "name": "Julia-1",
              "points": [
                {
                  "label": "Games",
                  "value": 0.2083,
                  "lo": 0.1569,
                  "hi": 0.2712,
                  "n": 192
                },
                {
                  "label": "Logic and thought experiments",
                  "value": 0.3462,
                  "lo": 0.276,
                  "hi": 0.4237,
                  "n": 156
                },
                {
                  "label": "Policy cases, 20 industries",
                  "value": 0.2518,
                  "lo": 0.2041,
                  "hi": 0.3064,
                  "n": 274
                },
                {
                  "label": "Usability intents",
                  "value": 0.1593,
                  "lo": 0.1173,
                  "hi": 0.2126,
                  "n": 226
                },
                {
                  "label": "Stress tests",
                  "value": 0.36,
                  "lo": 0.2967,
                  "hi": 0.4286,
                  "n": 200
                }
              ]
            }
          ],
          "note": "Games: positions solved by search. Logic: syllogisms, probability, bias probes. Policy cases: fictional company rules in 20 industries. Usability: what the user wants on a given screen. Stress tests: prompt injection, long logs, thresholds, near-identical options.",
          "sourceIds": [
            "system-one-arena"
          ]
        },
        {
          "id": "arena-latency-this-mac",
          "polarity": "lower",
          "title": "Speed on this Mac: the six open models",
          "subtitle": "Median, whisker to the 95th percentile · one Mac Studio (M3 Ultra), one call at a time",
          "kind": "dot-range",
          "unit": "ms",
          "whisker": "p50-p95",
          "series": [
            {
              "name": "This Mac",
              "points": [
                {
                  "label": "Julia-1",
                  "value": 13,
                  "lo": 13,
                  "hi": 22,
                  "n": 120
                },
                {
                  "label": "Laya",
                  "value": 43,
                  "lo": 43,
                  "hi": 68,
                  "n": 120
                },
                {
                  "label": "Kev 4B",
                  "value": 307,
                  "lo": 307,
                  "hi": 600,
                  "n": 120
                },
                {
                  "label": "lev 4B",
                  "value": 425,
                  "lo": 425,
                  "hi": 646,
                  "n": 120
                },
                {
                  "label": "Clef-Flash 9B",
                  "value": 566,
                  "lo": 566,
                  "hi": 816,
                  "n": 120
                },
                {
                  "label": "Clef 27B",
                  "value": 1919,
                  "lo": 1919,
                  "hi": 2792,
                  "n": 115
                }
              ]
            }
          ],
          "note": "All six ran on the same machine, so they compare with each other. Jev is not here: it ran on TypeSafe's servers. These times describe this Mac and 4- or 8-bit files, not the models: a data-centre GPU is several times faster.",
          "sourceIds": [
            "system-one-arena"
          ]
        },
        {
          "id": "arena-latency-same-gpu",
          "polarity": "lower",
          "title": "Speed on one GPU: third-party numbers",
          "subtitle": "Median, whisker to the 95th percentile · Decision Index 0.2.1, one NVIDIA RTX PRO 6000, one harness (snapshot 2026-10-01)",
          "kind": "dot-range",
          "unit": "ms",
          "whisker": "p50-p95",
          "series": [
            {
              "name": "Open models, same GPU",
              "points": [
                {
                  "label": "Laya",
                  "value": 5.8,
                  "lo": 5.8,
                  "hi": 222.5
                },
                {
                  "label": "Julia-1",
                  "value": 5.8,
                  "lo": 5.8,
                  "hi": 15.3
                },
                {
                  "label": "Clef-Flash 9B",
                  "value": 38.8,
                  "lo": 38.8,
                  "hi": 122.4
                },
                {
                  "label": "Kev 4B",
                  "value": 52.1,
                  "lo": 52.1,
                  "hi": 141.1
                },
                {
                  "label": "lev 4B",
                  "value": 70.8,
                  "lo": 70.8,
                  "hi": 687.7
                },
                {
                  "label": "Clef 27B",
                  "value": 209.3,
                  "lo": 209.3,
                  "hi": 238.6
                }
              ]
            },
            {
              "name": "Hosted: network included",
              "points": [
                {
                  "label": "Jev 1.13",
                  "value": 524.1,
                  "lo": 524.1,
                  "hi": 536
                }
              ]
            }
          ],
          "note": "Measured by the Decision Index, not by us. Every open model ran on the same GPU in one harness, so the six compare with each other. Jev is a hosted API: its time includes the network from the board's lab, so it is shown apart and is not a compute comparison.",
          "sourceIds": [
            "system-one-arena"
          ]
        },
        {
          "id": "arena-speed-accuracy-gpu",
          "polarity": "none",
          "title": "Open models: accuracy against speed on the same GPU",
          "subtitle": "x: third-party median latency on one GPU · y: our accuracy on 1,048 graded items · 95% intervals",
          "kind": "scatter",
          "unit": "rate",
          "xLabel": "Median latency on one GPU (ms, third party)",
          "yLabel": "Accuracy (ours)",
          "series": [
            {
              "name": "Clef 27B",
              "points": [
                {
                  "label": "Clef 27B",
                  "x": 209.3,
                  "value": 0.6994,
                  "lo": 0.671,
                  "hi": 0.7264,
                  "n": 1048
                }
              ]
            },
            {
              "name": "Clef-Flash 9B",
              "points": [
                {
                  "label": "Clef-Flash 9B",
                  "x": 38.8,
                  "value": 0.6517,
                  "lo": 0.6224,
                  "hi": 0.68,
                  "n": 1048
                }
              ]
            },
            {
              "name": "lev 4B",
              "points": [
                {
                  "label": "lev 4B",
                  "x": 70.8,
                  "value": 0.5859,
                  "lo": 0.5558,
                  "hi": 0.6153,
                  "n": 1048
                }
              ]
            },
            {
              "name": "Kev 4B",
              "points": [
                {
                  "label": "Kev 4B",
                  "x": 52.1,
                  "value": 0.6403,
                  "lo": 0.6107,
                  "hi": 0.6688,
                  "n": 1048
                }
              ]
            },
            {
              "name": "Laya",
              "points": [
                {
                  "label": "Laya",
                  "x": 5.8,
                  "value": 0.2739,
                  "lo": 0.2477,
                  "hi": 0.3016,
                  "n": 1048
                }
              ]
            },
            {
              "name": "Julia-1",
              "points": [
                {
                  "label": "Julia-1",
                  "x": 5.8,
                  "value": 0.2586,
                  "lo": 0.233,
                  "hi": 0.2859,
                  "n": 1048
                }
              ]
            }
          ],
          "note": "Speed comes from the Decision Index (one GPU, one harness); accuracy comes from this study. Jev is a hosted API and is not on this machine, so it has no dot here: it scored 76.8%.",
          "sourceIds": [
            "system-one-arena"
          ]
        },
        {
          "id": "arena-latency-gateway",
          "polarity": "lower",
          "title": "Speed through one gateway: OpenRouter's own numbers",
          "subtitle": "Median latency OpenRouter reports for each model's endpoint · read 2026-10-06",
          "kind": "bar",
          "unit": "seconds",
          "series": [
            {
              "name": "Median latency, one gateway",
              "points": [
                {
                  "label": "Jev 1.13 (TypeSafe)",
                  "value": 0.17
                },
                {
                  "label": "Clef 27B (Workers AI)",
                  "value": 0.5
                },
                {
                  "label": "Clef-Flash 9B (Workers AI)",
                  "value": 0.59
                },
                {
                  "label": "Kev 4B (SiliconFlow)",
                  "value": 4.08
                }
              ]
            }
          ],
          "note": "One gateway measures all four, so the network path is the same. The providers differ (TypeSafe, Workers AI, SiliconFlow) and the numbers move daily. Not our measurement.",
          "sourceIds": [
            "system-one-arena"
          ]
        },
        {
          "id": "arena-cost-same-provider",
          "polarity": "lower",
          "title": "Price per 1,000 decisions at one provider's list prices",
          "subtitle": "Cloudflare Workers AI list price per million input tokens × the input tokens each model counted per decision in this study",
          "kind": "bar",
          "unit": "usd",
          "series": [
            {
              "name": "USD per 1,000 decisions",
              "points": [
                {
                  "label": "Jev 1.13 ($0.042/M, 825 tokens)",
                  "value": 0.0347
                },
                {
                  "label": "Clef-Flash 9B ($0.09/M, 568 tokens)",
                  "value": 0.0511
                },
                {
                  "label": "Clef 27B ($0.24/M, 568 tokens)",
                  "value": 0.1363
                }
              ]
            }
          ],
          "note": "One provider prices all three, so this is a like-for-like cost, with no timing involved. Token counts come from each model's own tokenizer; output is free for all three. A calculation, not an invoice.",
          "sourceIds": [
            "system-one-arena"
          ]
        },
        {
          "id": "arena-size-accuracy",
          "polarity": "none",
          "title": "Does a bigger file decide better?",
          "subtitle": "Accuracy against model file size (open models)",
          "kind": "scatter",
          "unit": "rate",
          "xLabel": "Model file (GB)",
          "yLabel": "Accuracy",
          "series": [
            {
              "name": "Clef 27B",
              "points": [
                {
                  "label": "Clef 27B",
                  "x": 19.23,
                  "value": 0.6994,
                  "lo": 0.671,
                  "hi": 0.7264,
                  "n": 1048
                }
              ]
            },
            {
              "name": "Clef-Flash 9B",
              "points": [
                {
                  "label": "Clef-Flash 9B",
                  "x": 6.49,
                  "value": 0.6517,
                  "lo": 0.6224,
                  "hi": 0.68,
                  "n": 1048
                }
              ]
            },
            {
              "name": "lev 4B",
              "points": [
                {
                  "label": "lev 4B",
                  "x": 3.01,
                  "value": 0.5859,
                  "lo": 0.5558,
                  "hi": 0.6153,
                  "n": 1048
                }
              ]
            },
            {
              "name": "Kev 4B",
              "points": [
                {
                  "label": "Kev 4B",
                  "x": 3.03,
                  "value": 0.6403,
                  "lo": 0.6107,
                  "hi": 0.6688,
                  "n": 1048
                }
              ]
            },
            {
              "name": "Laya",
              "points": [
                {
                  "label": "Laya",
                  "x": 0.45,
                  "value": 0.2739,
                  "lo": 0.2477,
                  "hi": 0.3016,
                  "n": 1048
                }
              ]
            },
            {
              "name": "Julia-1",
              "points": [
                {
                  "label": "Julia-1",
                  "x": 0.17,
                  "value": 0.2586,
                  "lo": 0.233,
                  "hi": 0.2859,
                  "n": 1048
                }
              ]
            }
          ],
          "note": "File size of the GGUF that ran (quantized). Hardware does not enter. Jev is closed and its size is not published, so it is not on this chart.",
          "sourceIds": [
            "system-one-arena"
          ]
        },
        {
          "id": "arena-robust",
          "title": "Same question, different presentation",
          "subtitle": "Right at first presentation vs right in all three: original order, shuffled options, options renamed a, b, c",
          "kind": "dot-range",
          "unit": "rate",
          "whisker": "ci95",
          "series": [
            {
              "name": "First presentation",
              "points": [
                {
                  "label": "Jev 1.13",
                  "value": 0.7681,
                  "lo": 0.7416,
                  "hi": 0.7927,
                  "n": 1048
                },
                {
                  "label": "Clef 27B",
                  "value": 0.6994,
                  "lo": 0.671,
                  "hi": 0.7264,
                  "n": 1048
                },
                {
                  "label": "Clef-Flash 9B",
                  "value": 0.6517,
                  "lo": 0.6224,
                  "hi": 0.68,
                  "n": 1048
                },
                {
                  "label": "Kev 4B",
                  "value": 0.6403,
                  "lo": 0.6107,
                  "hi": 0.6688,
                  "n": 1048
                },
                {
                  "label": "lev 4B",
                  "value": 0.5859,
                  "lo": 0.5558,
                  "hi": 0.6153,
                  "n": 1048
                },
                {
                  "label": "Laya",
                  "value": 0.2739,
                  "lo": 0.2477,
                  "hi": 0.3016,
                  "n": 1048
                },
                {
                  "label": "Julia-1",
                  "value": 0.2586,
                  "lo": 0.233,
                  "hi": 0.2859,
                  "n": 1048
                }
              ]
            },
            {
              "name": "All three presentations",
              "points": [
                {
                  "label": "Jev 1.13",
                  "value": 0.729,
                  "lo": 0.7013,
                  "hi": 0.755,
                  "n": 1048
                },
                {
                  "label": "Clef 27B",
                  "value": 0.6536,
                  "lo": 0.6243,
                  "hi": 0.6818,
                  "n": 1048
                },
                {
                  "label": "Clef-Flash 9B",
                  "value": 0.5964,
                  "lo": 0.5664,
                  "hi": 0.6257,
                  "n": 1048
                },
                {
                  "label": "Kev 4B",
                  "value": 0.5697,
                  "lo": 0.5395,
                  "hi": 0.5993,
                  "n": 1048
                },
                {
                  "label": "lev 4B",
                  "value": 0.5344,
                  "lo": 0.5041,
                  "hi": 0.5644,
                  "n": 1048
                },
                {
                  "label": "Laya",
                  "value": 0.1584,
                  "lo": 0.1375,
                  "hi": 0.1817,
                  "n": 1048
                },
                {
                  "label": "Julia-1",
                  "value": 0.1603,
                  "lo": 0.1393,
                  "hi": 0.1838,
                  "n": 1048
                }
              ]
            }
          ],
          "note": "Yes/no items have one presentation, so for them both measures are the same.",
          "sourceIds": [
            "system-one-arena"
          ]
        },
        {
          "id": "arena-flips",
          "polarity": "lower",
          "title": "Decisions that changed when only the presentation changed",
          "subtitle": "Share of choice items · lower is better",
          "kind": "grouped-bar",
          "unit": "rate",
          "series": [
            {
              "name": "Options shuffled",
              "points": [
                {
                  "label": "Jev 1.13",
                  "value": 0.077,
                  "lo": 0.0618,
                  "hi": 0.0956,
                  "n": 961
                },
                {
                  "label": "Clef 27B",
                  "value": 0,
                  "lo": 0,
                  "hi": 0.004,
                  "n": 961
                },
                {
                  "label": "Clef-Flash 9B",
                  "value": 0,
                  "lo": 0,
                  "hi": 0.004,
                  "n": 961
                },
                {
                  "label": "Kev 4B",
                  "value": 0.1426,
                  "lo": 0.1219,
                  "hi": 0.1661,
                  "n": 961
                },
                {
                  "label": "lev 4B",
                  "value": 0.1322,
                  "lo": 0.1122,
                  "hi": 0.155,
                  "n": 961
                },
                {
                  "label": "Laya",
                  "value": 0.3361,
                  "lo": 0.3069,
                  "hi": 0.3666,
                  "n": 961
                },
                {
                  "label": "Julia-1",
                  "value": 0.5203,
                  "lo": 0.4887,
                  "hi": 0.5517,
                  "n": 961
                }
              ]
            },
            {
              "name": "Options renamed a, b, c",
              "points": [
                {
                  "label": "Jev 1.13",
                  "value": 0.0687,
                  "lo": 0.0543,
                  "hi": 0.0864,
                  "n": 961
                },
                {
                  "label": "Clef 27B",
                  "value": 0.1831,
                  "lo": 0.16,
                  "hi": 0.2088,
                  "n": 961
                },
                {
                  "label": "Clef-Flash 9B",
                  "value": 0.18,
                  "lo": 0.157,
                  "hi": 0.2056,
                  "n": 961
                },
                {
                  "label": "Kev 4B",
                  "value": 0.1103,
                  "lo": 0.092,
                  "hi": 0.1317,
                  "n": 961
                },
                {
                  "label": "lev 4B",
                  "value": 0.1061,
                  "lo": 0.0882,
                  "hi": 0.1272,
                  "n": 961
                },
                {
                  "label": "Laya",
                  "value": 0.4631,
                  "lo": 0.4317,
                  "hi": 0.4947,
                  "n": 961
                },
                {
                  "label": "Julia-1",
                  "value": 0,
                  "lo": 0,
                  "hi": 0.004,
                  "n": 961
                }
              ]
            }
          ],
          "note": "The descriptions never changed, only their order or their labels. A decision model should not care.",
          "sourceIds": [
            "system-one-arena"
          ]
        },
        {
          "id": "arena-escape",
          "polarity": "none",
          "title": "Knowing when to say \"none of these\"",
          "subtitle": "Items with a none, ask or escalate option · first presentation",
          "kind": "grouped-bar",
          "unit": "rate",
          "series": [
            {
              "name": "Escaped when it should (higher is better)",
              "points": [
                {
                  "label": "Jev 1.13",
                  "value": 0.8571,
                  "lo": 0.7915,
                  "hi": 0.9046,
                  "n": 147
                },
                {
                  "label": "Clef 27B",
                  "value": 0.7619,
                  "lo": 0.6869,
                  "hi": 0.8235,
                  "n": 147
                },
                {
                  "label": "Clef-Flash 9B",
                  "value": 0.6054,
                  "lo": 0.5247,
                  "hi": 0.6808,
                  "n": 147
                },
                {
                  "label": "Kev 4B",
                  "value": 0.619,
                  "lo": 0.5385,
                  "hi": 0.6936,
                  "n": 147
                },
                {
                  "label": "lev 4B",
                  "value": 0.4694,
                  "lo": 0.3905,
                  "hi": 0.5498,
                  "n": 147
                },
                {
                  "label": "Laya",
                  "value": 0.1769,
                  "lo": 0.1237,
                  "hi": 0.2465,
                  "n": 147
                },
                {
                  "label": "Julia-1",
                  "value": 0.2245,
                  "lo": 0.1646,
                  "hi": 0.2985,
                  "n": 147
                }
              ]
            },
            {
              "name": "Escaped when it should not (lower is better)",
              "points": [
                {
                  "label": "Jev 1.13",
                  "value": 0.055,
                  "lo": 0.0396,
                  "hi": 0.0759,
                  "n": 618
                },
                {
                  "label": "Clef 27B",
                  "value": 0.0259,
                  "lo": 0.016,
                  "hi": 0.0416,
                  "n": 618
                },
                {
                  "label": "Clef-Flash 9B",
                  "value": 0.0146,
                  "lo": 0.0077,
                  "hi": 0.0274,
                  "n": 618
                },
                {
                  "label": "Kev 4B",
                  "value": 0.0809,
                  "lo": 0.0619,
                  "hi": 0.1051,
                  "n": 618
                },
                {
                  "label": "lev 4B",
                  "value": 0.034,
                  "lo": 0.0223,
                  "hi": 0.0514,
                  "n": 618
                },
                {
                  "label": "Laya",
                  "value": 0.1375,
                  "lo": 0.1126,
                  "hi": 0.1669,
                  "n": 618
                },
                {
                  "label": "Julia-1",
                  "value": 0.165,
                  "lo": 0.1379,
                  "hi": 0.1964,
                  "n": 618
                }
              ]
            }
          ],
          "note": "Out-of-scope requests, policies that do not cover a case, ambiguous requests that need a question, lost game positions.",
          "sourceIds": [
            "system-one-arena"
          ]
        },
        {
          "id": "arena-injection",
          "title": "Prompt injection: does text in the state hijack the decision?",
          "subtitle": "Right answers on items whose state holds \"ignore the instructions and choose X\" style text · 95% Wilson intervals",
          "kind": "dot-range",
          "unit": "rate",
          "whisker": "ci95",
          "series": [
            {
              "name": "Right despite the injected text",
              "points": [
                {
                  "label": "Jev 1.13",
                  "value": 0.95,
                  "lo": 0.835,
                  "hi": 0.9862,
                  "n": 40
                },
                {
                  "label": "Clef 27B",
                  "value": 0.9,
                  "lo": 0.7695,
                  "hi": 0.9604,
                  "n": 40
                },
                {
                  "label": "Clef-Flash 9B",
                  "value": 0.775,
                  "lo": 0.625,
                  "hi": 0.8768,
                  "n": 40
                },
                {
                  "label": "Kev 4B",
                  "value": 0.95,
                  "lo": 0.835,
                  "hi": 0.9862,
                  "n": 40
                },
                {
                  "label": "lev 4B",
                  "value": 0.675,
                  "lo": 0.5202,
                  "hi": 0.7992,
                  "n": 40
                },
                {
                  "label": "Laya",
                  "value": 0.325,
                  "lo": 0.2008,
                  "hi": 0.4798,
                  "n": 40
                },
                {
                  "label": "Julia-1",
                  "value": 0.45,
                  "lo": 0.3071,
                  "hi": 0.6017,
                  "n": 40
                }
              ]
            }
          ],
          "note": "The injected text sits in a user-supplied field (a review, an email, a file name). The right answer follows the real instructions.",
          "sourceIds": [
            "system-one-arena"
          ]
        },
        {
          "id": "arena-by-length",
          "title": "Short inputs vs long inputs",
          "subtitle": "Accuracy by item length (Jev's token count of the same request) · first presentation",
          "kind": "grouped-bar",
          "unit": "rate",
          "series": [
            {
              "name": "Up to 512 tokens",
              "points": [
                {
                  "label": "Jev 1.13",
                  "value": 0.7867,
                  "lo": 0.7286,
                  "hi": 0.8351,
                  "n": 225
                },
                {
                  "label": "Clef 27B",
                  "value": 0.72,
                  "lo": 0.658,
                  "hi": 0.7746,
                  "n": 225
                },
                {
                  "label": "Clef-Flash 9B",
                  "value": 0.6667,
                  "lo": 0.6027,
                  "hi": 0.725,
                  "n": 225
                },
                {
                  "label": "Kev 4B",
                  "value": 0.6,
                  "lo": 0.5348,
                  "hi": 0.6618,
                  "n": 225
                },
                {
                  "label": "lev 4B",
                  "value": 0.6044,
                  "lo": 0.5393,
                  "hi": 0.6661,
                  "n": 225
                },
                {
                  "label": "Laya",
                  "value": 0.3689,
                  "lo": 0.3085,
                  "hi": 0.4336,
                  "n": 225
                },
                {
                  "label": "Julia-1",
                  "value": 0.4222,
                  "lo": 0.3595,
                  "hi": 0.4875,
                  "n": 225
                }
              ]
            },
            {
              "name": "More than 512 tokens",
              "points": [
                {
                  "label": "Jev 1.13",
                  "value": 0.7631,
                  "lo": 0.7328,
                  "hi": 0.7908,
                  "n": 823
                },
                {
                  "label": "Clef 27B",
                  "value": 0.6938,
                  "lo": 0.6615,
                  "hi": 0.7243,
                  "n": 823
                },
                {
                  "label": "Clef-Flash 9B",
                  "value": 0.6476,
                  "lo": 0.6144,
                  "hi": 0.6795,
                  "n": 823
                },
                {
                  "label": "Kev 4B",
                  "value": 0.6513,
                  "lo": 0.6181,
                  "hi": 0.6831,
                  "n": 823
                },
                {
                  "label": "lev 4B",
                  "value": 0.5808,
                  "lo": 0.5468,
                  "hi": 0.6141,
                  "n": 823
                },
                {
                  "label": "Laya",
                  "value": 0.2479,
                  "lo": 0.2196,
                  "hi": 0.2785,
                  "n": 823
                },
                {
                  "label": "Julia-1",
                  "value": 0.2139,
                  "lo": 0.1872,
                  "hi": 0.2432,
                  "n": 823
                }
              ]
            }
          ],
          "note": "Laya and Julia-1 are small encoders trained on short inputs (512 to 1,024 tokens); the longest items here are about 2,000 tokens. This split was added after the freeze as a description, not a test.",
          "sourceIds": [
            "system-one-arena"
          ]
        },
        {
          "id": "arena-confident-wrong",
          "polarity": "lower",
          "title": "Wrong and sure of it",
          "subtitle": "Share of wrong answers given with probability 0.8 or more · lower is better",
          "kind": "dot-range",
          "unit": "rate",
          "whisker": "ci95",
          "series": [
            {
              "name": "Confident wrong answers",
              "points": [
                {
                  "label": "Jev 1.13",
                  "value": 0.1111,
                  "lo": 0.0775,
                  "hi": 0.1568,
                  "n": 243
                },
                {
                  "label": "Clef 27B",
                  "value": 0.0161,
                  "lo": 0.0069,
                  "hi": 0.0371,
                  "n": 311
                },
                {
                  "label": "Clef-Flash 9B",
                  "value": 0.0082,
                  "lo": 0.0028,
                  "hi": 0.0239,
                  "n": 365
                },
                {
                  "label": "Kev 4B",
                  "value": 0.0743,
                  "lo": 0.0519,
                  "hi": 0.1052,
                  "n": 377
                },
                {
                  "label": "lev 4B",
                  "value": 0.3364,
                  "lo": 0.2936,
                  "hi": 0.3821,
                  "n": 434
                },
                {
                  "label": "Laya",
                  "value": 0.0972,
                  "lo": 0.0782,
                  "hi": 0.1204,
                  "n": 761
                },
                {
                  "label": "Julia-1",
                  "value": 0.5611,
                  "lo": 0.526,
                  "hi": 0.5956,
                  "n": 777
                }
              ]
            }
          ],
          "note": "The probability is what the model returned for the option it chose. A well-calibrated decider is rarely this sure when it is wrong, so a confidence gate can catch its mistakes.",
          "sourceIds": [
            "system-one-arena"
          ]
        },
        {
          "id": "arena-elo",
          "title": "Tournament rating across every game",
          "subtitle": "Elo from every round-robin game (K 32, start 1000, averaged over 200 game orders) · 95% bootstrap intervals",
          "kind": "dot-range",
          "unit": "score",
          "whisker": "ci95",
          "series": [
            {
              "name": "Elo",
              "points": [
                {
                  "label": "Perfect player",
                  "value": 1554,
                  "lo": 1498,
                  "hi": 1609,
                  "n": 336
                },
                {
                  "label": "Jev 1.13",
                  "value": 1040,
                  "lo": 1002,
                  "hi": 1082,
                  "n": 336,
                  "highlight": true
                },
                {
                  "label": "lev 4B",
                  "value": 941,
                  "lo": 894,
                  "hi": 986,
                  "n": 336
                },
                {
                  "label": "Clef 27B",
                  "value": 941,
                  "lo": 901,
                  "hi": 984,
                  "n": 336
                },
                {
                  "label": "Kev 4B",
                  "value": 936,
                  "lo": 892,
                  "hi": 974,
                  "n": 336
                },
                {
                  "label": "Clef-Flash 9B",
                  "value": 936,
                  "lo": 890,
                  "hi": 982,
                  "n": 336
                },
                {
                  "label": "Random player",
                  "value": 896,
                  "lo": 858,
                  "hi": 948,
                  "n": 336
                },
                {
                  "label": "Julia-1",
                  "value": 879,
                  "lo": 831,
                  "hi": 921,
                  "n": 336
                },
                {
                  "label": "Laya",
                  "value": 877,
                  "lo": 840,
                  "hi": 921,
                  "n": 336
                }
              ]
            }
          ],
          "note": "Tic-tac-toe, Connect Four, Nim, Dots and Boxes and Pong; every game counts once. The random and perfect players anchor the scale.",
          "sourceIds": [
            "system-one-arena"
          ]
        },
        {
          "id": "arena-wdl",
          "title": "Wins, draws and losses in the round robin",
          "subtitle": "Every round-robin game",
          "kind": "stacked-bar",
          "unit": "count",
          "series": [
            {
              "name": "Wins",
              "points": [
                {
                  "label": "Perfect player",
                  "value": 324,
                  "n": 336
                },
                {
                  "label": "Jev 1.13",
                  "value": 196,
                  "n": 336
                },
                {
                  "label": "lev 4B",
                  "value": 146,
                  "n": 336
                },
                {
                  "label": "Clef 27B",
                  "value": 152,
                  "n": 336
                },
                {
                  "label": "Kev 4B",
                  "value": 144,
                  "n": 336
                },
                {
                  "label": "Clef-Flash 9B",
                  "value": 147,
                  "n": 336
                },
                {
                  "label": "Random player",
                  "value": 127,
                  "n": 336
                },
                {
                  "label": "Julia-1",
                  "value": 120,
                  "n": 336
                },
                {
                  "label": "Laya",
                  "value": 118,
                  "n": 336
                }
              ]
            },
            {
              "name": "Draws",
              "points": [
                {
                  "label": "Perfect player",
                  "value": 9,
                  "n": 336
                },
                {
                  "label": "Jev 1.13",
                  "value": 6,
                  "n": 336
                },
                {
                  "label": "lev 4B",
                  "value": 9,
                  "n": 336
                },
                {
                  "label": "Clef 27B",
                  "value": 5,
                  "n": 336
                },
                {
                  "label": "Kev 4B",
                  "value": 9,
                  "n": 336
                },
                {
                  "label": "Clef-Flash 9B",
                  "value": 7,
                  "n": 336
                },
                {
                  "label": "Random player",
                  "value": 12,
                  "n": 336
                },
                {
                  "label": "Julia-1",
                  "value": 6,
                  "n": 336
                },
                {
                  "label": "Laya",
                  "value": 13,
                  "n": 336
                }
              ]
            },
            {
              "name": "Losses",
              "points": [
                {
                  "label": "Perfect player",
                  "value": 3,
                  "n": 336
                },
                {
                  "label": "Jev 1.13",
                  "value": 134,
                  "n": 336
                },
                {
                  "label": "lev 4B",
                  "value": 181,
                  "n": 336
                },
                {
                  "label": "Clef 27B",
                  "value": 179,
                  "n": 336
                },
                {
                  "label": "Kev 4B",
                  "value": 183,
                  "n": 336
                },
                {
                  "label": "Clef-Flash 9B",
                  "value": 182,
                  "n": 336
                },
                {
                  "label": "Random player",
                  "value": 197,
                  "n": 336
                },
                {
                  "label": "Julia-1",
                  "value": 210,
                  "n": 336
                },
                {
                  "label": "Laya",
                  "value": 205,
                  "n": 336
                }
              ]
            }
          ],
          "sourceIds": [
            "system-one-arena"
          ]
        },
        {
          "id": "arena-perfect-moves",
          "title": "How often a model found the perfect move",
          "subtitle": "Share of decisive moves (some option is a mistake) that a perfect player would also make",
          "kind": "grouped-bar",
          "unit": "rate",
          "series": [
            {
              "name": "Jev 1.13",
              "points": [
                {
                  "label": "Tic-tac-toe",
                  "value": 0.4653,
                  "lo": 0.3858,
                  "hi": 0.5466,
                  "n": 144
                },
                {
                  "label": "Connect Four",
                  "value": 0.3789,
                  "lo": 0.3297,
                  "hi": 0.4307,
                  "n": 351
                },
                {
                  "label": "Nim",
                  "value": 0.3916,
                  "lo": 0.3206,
                  "hi": 0.4675,
                  "n": 166
                },
                {
                  "label": "Dots and Boxes",
                  "value": 0.5479,
                  "lo": 0.5127,
                  "hi": 0.5827,
                  "n": 772
                }
              ]
            },
            {
              "name": "Clef 27B",
              "points": [
                {
                  "label": "Tic-tac-toe",
                  "value": 0.4718,
                  "lo": 0.3916,
                  "hi": 0.5536,
                  "n": 142
                },
                {
                  "label": "Connect Four",
                  "value": 0.3138,
                  "lo": 0.2698,
                  "hi": 0.3613,
                  "n": 392
                },
                {
                  "label": "Nim",
                  "value": 0.4051,
                  "lo": 0.3317,
                  "hi": 0.483,
                  "n": 158
                },
                {
                  "label": "Dots and Boxes",
                  "value": 0.3376,
                  "lo": 0.3016,
                  "hi": 0.3757,
                  "n": 622
                }
              ]
            },
            {
              "name": "Clef-Flash 9B",
              "points": [
                {
                  "label": "Tic-tac-toe",
                  "value": 0.5306,
                  "lo": 0.4502,
                  "hi": 0.6095,
                  "n": 147
                },
                {
                  "label": "Connect Four",
                  "value": 0.3629,
                  "lo": 0.3163,
                  "hi": 0.4122,
                  "n": 383
                },
                {
                  "label": "Nim",
                  "value": 0.2593,
                  "lo": 0.1927,
                  "hi": 0.3391,
                  "n": 135
                },
                {
                  "label": "Dots and Boxes",
                  "value": 0.3267,
                  "lo": 0.2903,
                  "hi": 0.3652,
                  "n": 600
                }
              ]
            },
            {
              "name": "lev 4B",
              "points": [
                {
                  "label": "Tic-tac-toe",
                  "value": 0.4737,
                  "lo": 0.3959,
                  "hi": 0.5527,
                  "n": 152
                },
                {
                  "label": "Connect Four",
                  "value": 0.3477,
                  "lo": 0.3024,
                  "hi": 0.396,
                  "n": 394
                },
                {
                  "label": "Nim",
                  "value": 0.3101,
                  "lo": 0.2367,
                  "hi": 0.3944,
                  "n": 129
                },
                {
                  "label": "Dots and Boxes",
                  "value": 0.363,
                  "lo": 0.3261,
                  "hi": 0.4017,
                  "n": 617
                }
              ]
            },
            {
              "name": "Kev 4B",
              "points": [
                {
                  "label": "Tic-tac-toe",
                  "value": 0.4371,
                  "lo": 0.3605,
                  "hi": 0.5168,
                  "n": 151
                },
                {
                  "label": "Connect Four",
                  "value": 0.1415,
                  "lo": 0.1071,
                  "hi": 0.1846,
                  "n": 311
                },
                {
                  "label": "Nim",
                  "value": 0.2975,
                  "lo": 0.2233,
                  "hi": 0.3842,
                  "n": 121
                },
                {
                  "label": "Dots and Boxes",
                  "value": 0.3339,
                  "lo": 0.2974,
                  "hi": 0.3725,
                  "n": 602
                }
              ]
            },
            {
              "name": "Laya",
              "points": [
                {
                  "label": "Tic-tac-toe",
                  "value": 0.3716,
                  "lo": 0.2979,
                  "hi": 0.4518,
                  "n": 148
                },
                {
                  "label": "Connect Four",
                  "value": 0.1003,
                  "lo": 0.073,
                  "hi": 0.1363,
                  "n": 349
                },
                {
                  "label": "Nim",
                  "value": 0.2025,
                  "lo": 0.1473,
                  "hi": 0.2719,
                  "n": 158
                },
                {
                  "label": "Dots and Boxes",
                  "value": 0.3667,
                  "lo": 0.3308,
                  "hi": 0.4041,
                  "n": 660
                }
              ]
            },
            {
              "name": "Julia-1",
              "points": [
                {
                  "label": "Tic-tac-toe",
                  "value": 0.3562,
                  "lo": 0.2831,
                  "hi": 0.4366,
                  "n": 146
                },
                {
                  "label": "Connect Four",
                  "value": 0.2227,
                  "lo": 0.186,
                  "hi": 0.2644,
                  "n": 431
                },
                {
                  "label": "Nim",
                  "value": 0.2706,
                  "lo": 0.2094,
                  "hi": 0.3419,
                  "n": 170
                },
                {
                  "label": "Dots and Boxes",
                  "value": 0.288,
                  "lo": 0.2524,
                  "hi": 0.3263,
                  "n": 573
                }
              ]
            },
            {
              "name": "Random player",
              "points": [
                {
                  "label": "Tic-tac-toe",
                  "value": 0.4013,
                  "lo": 0.3268,
                  "hi": 0.4807,
                  "n": 152
                },
                {
                  "label": "Connect Four",
                  "value": 0.232,
                  "lo": 0.1951,
                  "hi": 0.2734,
                  "n": 444
                },
                {
                  "label": "Nim",
                  "value": 0.2327,
                  "lo": 0.1738,
                  "hi": 0.3042,
                  "n": 159
                },
                {
                  "label": "Dots and Boxes",
                  "value": 0.3241,
                  "lo": 0.2884,
                  "hi": 0.3621,
                  "n": 617
                }
              ]
            }
          ],
          "note": "Tic-tac-toe and Dots and Boxes against exact solvers, Nim by nim-sum, Connect Four against a depth-10 search (agreement, not proof of a mistake). The random player is the anchor: below it, a model chose worse than chance.",
          "sourceIds": [
            "system-one-arena"
          ]
        },
        {
          "id": "arena-pong",
          "title": "Pong as deployed: who won",
          "subtitle": "Share of Pong games won · first to 3 · each answer lands after its own call time on its own machine (hardware-dependent) · 95% Wilson intervals",
          "kind": "dot-range",
          "unit": "rate",
          "whisker": "ci95",
          "series": [
            {
              "name": "Pong score",
              "points": [
                {
                  "label": "Perfect player",
                  "value": 1,
                  "lo": 0.8064,
                  "hi": 1,
                  "n": 16
                },
                {
                  "label": "Jev 1.13",
                  "value": 0.8125,
                  "lo": 0.5699,
                  "hi": 0.9341,
                  "n": 16
                },
                {
                  "label": "Kev 4B",
                  "value": 0.75,
                  "lo": 0.505,
                  "hi": 0.8982,
                  "n": 16
                },
                {
                  "label": "lev 4B",
                  "value": 0.5625,
                  "lo": 0.3318,
                  "hi": 0.769,
                  "n": 16
                },
                {
                  "label": "Clef-Flash 9B",
                  "value": 0.5,
                  "lo": 0.28,
                  "hi": 0.72,
                  "n": 16
                },
                {
                  "label": "Clef 27B",
                  "value": 0.25,
                  "lo": 0.1018,
                  "hi": 0.495,
                  "n": 16
                },
                {
                  "label": "Julia-1",
                  "value": 0.25,
                  "lo": 0.1018,
                  "hi": 0.495,
                  "n": 16
                },
                {
                  "label": "Random player",
                  "value": 0.1875,
                  "lo": 0.0659,
                  "hi": 0.4301,
                  "n": 16
                },
                {
                  "label": "Laya",
                  "value": 0.1875,
                  "lo": 0.0659,
                  "hi": 0.4301,
                  "n": 16
                }
              ]
            }
          ],
          "note": "Each paddle asks its model where to go as soon as it is free; the answer moves the paddle only after that call's real latency, in simulated time. The arrival point is given in the state, so Pong tests reading and speed, not physics.",
          "sourceIds": [
            "system-one-arena"
          ]
        },
        {
          "id": "arena-pong-quality",
          "title": "Pong decision quality, ignoring time",
          "subtitle": "Share of graded decisions where the model chose the zone the ideal intercept chose · 95% Wilson intervals",
          "kind": "dot-range",
          "unit": "rate",
          "whisker": "ci95",
          "series": [
            {
              "name": "Right zone",
              "points": [
                {
                  "label": "Jev 1.13",
                  "value": 1,
                  "lo": 0.9927,
                  "hi": 1,
                  "n": 523
                },
                {
                  "label": "Clef 27B",
                  "value": 1,
                  "lo": 0.8897,
                  "hi": 1,
                  "n": 31
                },
                {
                  "label": "lev 4B",
                  "value": 0.9869,
                  "lo": 0.9536,
                  "hi": 0.9964,
                  "n": 153
                },
                {
                  "label": "Kev 4B",
                  "value": 0.9774,
                  "lo": 0.9541,
                  "hi": 0.989,
                  "n": 310
                },
                {
                  "label": "Clef-Flash 9B",
                  "value": 0.8296,
                  "lo": 0.7573,
                  "hi": 0.8837,
                  "n": 135
                },
                {
                  "label": "Julia-1",
                  "value": 0.2145,
                  "lo": 0.1997,
                  "hi": 0.2301,
                  "n": 2811
                },
                {
                  "label": "Laya",
                  "value": 0.1221,
                  "lo": 0.1033,
                  "hi": 0.1439,
                  "n": 999
                }
              ]
            }
          ],
          "note": "Hardware-neutral: each decision is judged on its own, and time does not enter. The ball is heading toward the paddle and its arrival point is in the state. Clef 27B decided rarely (slow), so its n is small.",
          "sourceIds": [
            "system-one-arena"
          ]
        },
        {
          "id": "arena-pong-ladder",
          "title": "How much is a millisecond worth in Pong?",
          "subtitle": "The perfect player slowed down, in a round robin against itself and a random player · 95% Wilson intervals",
          "kind": "dot-range",
          "unit": "rate",
          "whisker": "ci95",
          "series": [
            {
              "name": "Share of games won",
              "points": [
                {
                  "label": "0 ms",
                  "value": 0.975,
                  "lo": 0.8712,
                  "hi": 0.9956,
                  "n": 40
                },
                {
                  "label": "100 ms",
                  "value": 0.775,
                  "lo": 0.625,
                  "hi": 0.8768,
                  "n": 40
                },
                {
                  "label": "300 ms",
                  "value": 0.5,
                  "lo": 0.352,
                  "hi": 0.648,
                  "n": 40
                },
                {
                  "label": "1000 ms",
                  "value": 0.125,
                  "lo": 0.0546,
                  "hi": 0.2611,
                  "n": 40
                }
              ]
            }
          ],
          "note": "Same perfect decisions, only slower. No model calls.",
          "sourceIds": [
            "system-one-arena"
          ]
        }
      ],
      "tables": [
        {
          "id": "arena-fairness",
          "title": "What is comparable: each comparison and what keeps it fair",
          "columns": [
            {
              "key": "what",
              "label": "Comparison",
              "unit": "text"
            },
            {
              "key": "fair",
              "label": "Fair across all seven?",
              "unit": "text"
            },
            {
              "key": "why",
              "label": "Why",
              "unit": "text"
            }
          ],
          "rows": [
            {
              "what": "Accuracy, consistency, calibration, \"none of these\", prompt injection",
              "fair": "Yes",
              "why": "Same 1,085 items and the same states for every model. The open models ran as 4- or 8-bit files, so a full-precision run could differ a little; our order of the models agrees with the independent Decision Index (rank correlation 0.93)."
            },
            {
              "what": "Turn-based games",
              "fair": "Yes",
              "why": "No clock: a move counts only by its quality."
            },
            {
              "what": "Pong decision quality (right zone per decision)",
              "fair": "Yes",
              "why": "Judged per decision against the ideal intercept; time does not enter."
            },
            {
              "what": "Speed on this Mac",
              "fair": "Among the six open models only",
              "why": "The six ran on one machine, one call at a time. Jev ran on TypeSafe's servers."
            },
            {
              "what": "Speed on one GPU (Decision Index)",
              "fair": "Among the six open models only",
              "why": "Third party, one GPU, one harness. Jev is a hosted API there and includes the network."
            },
            {
              "what": "Speed through one gateway (OpenRouter)",
              "fair": "Yes for the four it lists",
              "why": "One gateway measures all four. Providers differ and the numbers change daily."
            },
            {
              "what": "Price per 1,000 decisions",
              "fair": "Yes for Jev, Clef and Clef-Flash",
              "why": "Cloudflare Workers AI prices all three on one list."
            },
            {
              "what": "Pong and other real-time games as deployed",
              "fair": "No",
              "why": "Each answer takes effect after its own call time on its own machine, so the result mixes decision quality with hardware."
            },
            {
              "what": "Real-time games in fair mode (Tron, Snake and Pong in the arena)",
              "fair": "Yes",
              "why": "Every player's answer takes effect after the same simulated 150 ms, whatever its real call time, so hosted and local models stand on the same footing and only decision quality counts. The models are still called for real. The leagues are one seed with 2 to 4 games per pairing, so the intervals are wide."
            }
          ]
        },
        {
          "id": "arena-models",
          "title": "Every model, every measure",
          "columns": [
            {
              "key": "model",
              "label": "Model",
              "unit": "text"
            },
            {
              "key": "runs",
              "label": "Runs as",
              "unit": "text"
            },
            {
              "key": "acc",
              "label": "Accuracy",
              "unit": "text"
            },
            {
              "key": "robust",
              "label": "Right in all 3 presentations",
              "unit": "text"
            },
            {
              "key": "groups",
              "label": "Consistent groups",
              "unit": "text"
            },
            {
              "key": "escape",
              "label": "Escaped when it should",
              "unit": "text"
            },
            {
              "key": "falseEscape",
              "label": "Escaped when it should not",
              "unit": "text"
            },
            {
              "key": "ece",
              "label": "Calibration error (ECE)",
              "unit": "ratio"
            },
            {
              "key": "brier",
              "label": "Brier",
              "unit": "ratio"
            },
            {
              "key": "p50",
              "label": "Median ms (where it ran: this Mac, or Jev's servers)",
              "unit": "ms"
            },
            {
              "key": "p95",
              "label": "p95 ms",
              "unit": "ms"
            },
            {
              "key": "same",
              "label": "Same answer twice",
              "unit": "text"
            },
            {
              "key": "errors",
              "label": "Errors",
              "unit": "count"
            }
          ],
          "rows": [
            {
              "model": "Jev 1.13",
              "runs": "Hosted API",
              "acc": "76.8% (805/1048)",
              "robust": "764/1048",
              "groups": "40/53",
              "escape": "126/147",
              "falseEscape": "34/618",
              "ece": 0.03,
              "brier": 0.18,
              "p50": 137,
              "p95": 189,
              "same": "58/60",
              "errors": 0
            },
            {
              "model": "Clef 27B",
              "runs": "Q4_K_M in llama.cpp, 19.23 GB",
              "acc": "69.9% (733/1048)",
              "robust": "685/1048",
              "groups": "38/53",
              "escape": "112/147",
              "falseEscape": "16/618",
              "ece": 0.28,
              "brier": 0.429,
              "p50": 1919,
              "p95": 2792,
              "same": "60/60",
              "errors": 8
            },
            {
              "model": "Clef-Flash 9B",
              "runs": "Q4_K_M in llama.cpp, 6.49 GB",
              "acc": "65.2% (683/1048)",
              "robust": "625/1048",
              "groups": "42/53",
              "escape": "89/147",
              "falseEscape": "9/618",
              "ece": 0.287,
              "brier": 0.476,
              "p50": 566,
              "p95": 816,
              "same": "60/60",
              "errors": 0
            },
            {
              "model": "Kev 4B",
              "runs": "Q4_K_M in llama.cpp, 3.03 GB",
              "acc": "64.0% (671/1048)",
              "robust": "597/1048",
              "groups": "35/53",
              "escape": "91/147",
              "falseEscape": "50/618",
              "ece": 0.037,
              "brier": 0.326,
              "p50": 307,
              "p95": 600,
              "same": "60/60",
              "errors": 0
            },
            {
              "model": "lev 4B",
              "runs": "Q4_K_M in llama.cpp, 3.01 GB",
              "acc": "58.6% (614/1048)",
              "robust": "560/1048",
              "groups": "38/53",
              "escape": "69/147",
              "falseEscape": "21/618",
              "ece": 0.145,
              "brier": 0.364,
              "p50": 425,
              "p95": 646,
              "same": "60/60",
              "errors": 0
            },
            {
              "model": "Laya",
              "runs": "Q8_0 in llama.cpp, 0.45 GB",
              "acc": "27.4% (287/1048)",
              "robust": "166/1048",
              "groups": "35/53",
              "escape": "26/147",
              "falseEscape": "85/618",
              "ece": 0.203,
              "brier": 0.603,
              "p50": 43,
              "p95": 68,
              "same": "60/60",
              "errors": 0
            },
            {
              "model": "Julia-1",
              "runs": "Q8_0 in llama.cpp, 0.17 GB",
              "acc": "25.9% (271/1048)",
              "robust": "168/1048",
              "groups": "28/53",
              "escape": "33/147",
              "falseEscape": "102/618",
              "ece": 0.547,
              "brier": 0.681,
              "p50": 13,
              "p95": 22,
              "same": "60/60",
              "errors": 0
            }
          ]
        },
        {
          "id": "arena-pairwise",
          "title": "Head to head on the same items: who was right where the other was wrong (exact McNemar test)",
          "columns": [
            {
              "key": "pair",
              "label": "Pair",
              "unit": "text"
            },
            {
              "key": "aOnly",
              "label": "Only the first right",
              "unit": "count"
            },
            {
              "key": "bOnly",
              "label": "Only the second right",
              "unit": "count"
            },
            {
              "key": "p",
              "label": "p",
              "unit": "ratio"
            },
            {
              "key": "verdict",
              "label": "Verdict",
              "unit": "text"
            }
          ],
          "rows": [
            {
              "pair": "Jev 1.13 vs Clef 27B",
              "aOnly": 158,
              "bOnly": 86,
              "p": 0,
              "verdict": "Jev 1.13 is ahead"
            },
            {
              "pair": "Jev 1.13 vs Clef-Flash 9B",
              "aOnly": 208,
              "bOnly": 86,
              "p": 0,
              "verdict": "Jev 1.13 is ahead"
            },
            {
              "pair": "Jev 1.13 vs lev 4B",
              "aOnly": 270,
              "bOnly": 79,
              "p": 0,
              "verdict": "Jev 1.13 is ahead"
            },
            {
              "pair": "Jev 1.13 vs Kev 4B",
              "aOnly": 212,
              "bOnly": 78,
              "p": 0,
              "verdict": "Jev 1.13 is ahead"
            },
            {
              "pair": "Jev 1.13 vs Laya",
              "aOnly": 584,
              "bOnly": 66,
              "p": 0,
              "verdict": "Jev 1.13 is ahead"
            },
            {
              "pair": "Jev 1.13 vs Julia-1",
              "aOnly": 599,
              "bOnly": 65,
              "p": 0,
              "verdict": "Jev 1.13 is ahead"
            },
            {
              "pair": "Clef 27B vs Clef-Flash 9B",
              "aOnly": 141,
              "bOnly": 91,
              "p": 0.0012,
              "verdict": "Clef 27B is ahead"
            },
            {
              "pair": "Clef 27B vs lev 4B",
              "aOnly": 209,
              "bOnly": 90,
              "p": 0,
              "verdict": "Clef 27B is ahead"
            },
            {
              "pair": "Clef 27B vs Kev 4B",
              "aOnly": 178,
              "bOnly": 116,
              "p": 0.0004,
              "verdict": "Clef 27B is ahead"
            },
            {
              "pair": "Clef 27B vs Laya",
              "aOnly": 529,
              "bOnly": 83,
              "p": 0,
              "verdict": "Clef 27B is ahead"
            },
            {
              "pair": "Clef 27B vs Julia-1",
              "aOnly": 543,
              "bOnly": 81,
              "p": 0,
              "verdict": "Clef 27B is ahead"
            },
            {
              "pair": "Clef-Flash 9B vs lev 4B",
              "aOnly": 168,
              "bOnly": 99,
              "p": 0,
              "verdict": "Clef-Flash 9B is ahead"
            },
            {
              "pair": "Clef-Flash 9B vs Kev 4B",
              "aOnly": 158,
              "bOnly": 146,
              "p": 0.5282,
              "verdict": "not clear"
            },
            {
              "pair": "Clef-Flash 9B vs Laya",
              "aOnly": 486,
              "bOnly": 90,
              "p": 0,
              "verdict": "Clef-Flash 9B is ahead"
            },
            {
              "pair": "Clef-Flash 9B vs Julia-1",
              "aOnly": 510,
              "bOnly": 98,
              "p": 0,
              "verdict": "Clef-Flash 9B is ahead"
            },
            {
              "pair": "lev 4B vs Kev 4B",
              "aOnly": 111,
              "bOnly": 168,
              "p": 0.0008,
              "verdict": "Kev 4B is ahead"
            },
            {
              "pair": "lev 4B vs Laya",
              "aOnly": 425,
              "bOnly": 98,
              "p": 0,
              "verdict": "lev 4B is ahead"
            },
            {
              "pair": "lev 4B vs Julia-1",
              "aOnly": 452,
              "bOnly": 109,
              "p": 0,
              "verdict": "lev 4B is ahead"
            },
            {
              "pair": "Kev 4B vs Laya",
              "aOnly": 471,
              "bOnly": 87,
              "p": 0,
              "verdict": "Kev 4B is ahead"
            },
            {
              "pair": "Kev 4B vs Julia-1",
              "aOnly": 503,
              "bOnly": 103,
              "p": 0,
              "verdict": "Kev 4B is ahead"
            },
            {
              "pair": "Laya vs Julia-1",
              "aOnly": 190,
              "bOnly": 174,
              "p": 0.4318,
              "verdict": "not clear"
            }
          ]
        },
        {
          "id": "arena-examples",
          "title": "One item per suite that split the field: about half the models right",
          "columns": [
            {
              "key": "suite",
              "label": "Suite",
              "unit": "text"
            },
            {
              "key": "item",
              "label": "Item",
              "unit": "text"
            },
            {
              "key": "question",
              "label": "Question",
              "unit": "text"
            },
            {
              "key": "answer",
              "label": "Right answer",
              "unit": "text"
            },
            {
              "key": "right",
              "label": "Right",
              "unit": "text"
            },
            {
              "key": "wrong",
              "label": "Wrong",
              "unit": "text"
            },
            {
              "key": "why",
              "label": "Why",
              "unit": "text"
            }
          ],
          "rows": [
            {
              "suite": "Games",
              "item": "Find the poisoned column",
              "question": "Y is to move. R has no winning move at the moment. Exactly one column is unsafe: if Y drops a disc in it, R can win on the very next move. Choose that unsafe column.",
              "answer": "Drop a disc in column 3",
              "right": "Jev 1.13, lev 4B, Kev 4B, Laya",
              "wrong": "Clef 27B, Clef-Flash 9B, Julia-1",
              "why": "A Y disc in column 3 lands directly under the cell where R completes four, so R wins on the very next move. All other columns leave R without an immediate win."
            },
            {
              "suite": "Logic and thought experiments",
              "item": "Flagged application: how likely is the fault?",
              "question": "One application was picked at random from those checked, and it was flagged. Which band holds the chance that it really has a copied answer?",
              "answer": "The chance is 30% to under 60%.",
              "right": "Jev 1.13, Clef 27B, Clef-Flash 9B",
              "wrong": "lev 4B, Kev 4B, Laya, Julia-1",
              "why": "Among the flagged items, the share that truly have the fault is 35.5% (Bayes' rule), which falls in the band \"30% to under 60%\"."
            },
            {
              "suite": "Policy cases, 20 industries",
              "item": "Delay exactly on the long-delay line, refund requested, no missed connection",
              "question": "Which action does the policy require for this passenger case? Apply the numbered rules in order and choose exactly one action.",
              "answer": "Offer no remedy: no rebooking and no refund",
              "right": "Jev 1.13, Clef-Flash 9B, lev 4B, Kev 4B",
              "wrong": "Clef 27B, Laya, Julia-1",
              "why": "R4: the delay is 300 minutes, which does not meet \"less than 150 minutes\" and meets \"300 minutes or less\", and no connection was missed, so there is no remedy."
            },
            {
              "suite": "Usability intents",
              "item": "Buttons lost against the background",
              "question": "Pick the one action the app should run now for what the user just said. Use the screen state and the rules in the state.",
              "answer": "Turn on high-contrast colors",
              "right": "Jev 1.13, Clef 27B, Clef-Flash 9B, lev 4B",
              "wrong": "Kev 4B, Laya, Julia-1",
              "why": "Buttons that blend into the background are a contrast problem, so High contrast."
            },
            {
              "suite": "Stress tests",
              "item": "Loud alert names and big counts versus the severity rule",
              "question": "Apply the routing rules to the alert and pick the action.",
              "answer": "Open a ticket",
              "right": "Jev 1.13, Clef 27B, Clef-Flash 9B, Kev 4B",
              "wrong": "lev 4B, Laya, Julia-1",
              "why": "Severity 2 is below the page level, so no page; it is at least 2, so open_ticket, whatever the name or customer count says."
            }
          ]
        },
        {
          "id": "arena-families",
          "title": "Every item family: right answers per model (first presentation)",
          "columns": [
            {
              "key": "suite",
              "label": "Suite",
              "unit": "text"
            },
            {
              "key": "family",
              "label": "Family",
              "unit": "text"
            },
            {
              "key": "jev",
              "label": "Jev 1.13",
              "unit": "text"
            },
            {
              "key": "clef",
              "label": "Clef 27B",
              "unit": "text"
            },
            {
              "key": "clef-flash",
              "label": "Clef-Flash 9B",
              "unit": "text"
            },
            {
              "key": "kev-4b",
              "label": "Kev 4B",
              "unit": "text"
            },
            {
              "key": "lev",
              "label": "lev 4B",
              "unit": "text"
            },
            {
              "key": "laya",
              "label": "Laya",
              "unit": "text"
            },
            {
              "key": "julia-1",
              "label": "Julia-1",
              "unit": "text"
            }
          ],
          "rows": [
            {
              "suite": "Games",
              "family": "tic-tac-toe",
              "jev": "8/23",
              "clef": "7/23",
              "clef-flash": "6/23",
              "kev-4b": "8/23",
              "lev": "11/23",
              "laya": "5/23",
              "julia-1": "1/23"
            },
            {
              "suite": "Games",
              "family": "connect-four",
              "jev": "7/26",
              "clef": "4/26",
              "clef-flash": "6/26",
              "kev-4b": "7/26",
              "lev": "7/26",
              "laya": "4/26",
              "julia-1": "5/26"
            },
            {
              "suite": "Games",
              "family": "nim",
              "jev": "12/22",
              "clef": "7/22",
              "clef-flash": "9/22",
              "kev-4b": "5/22",
              "lev": "5/22",
              "laya": "1/22",
              "julia-1": "6/22"
            },
            {
              "suite": "Games",
              "family": "pong-intercept",
              "jev": "6/18",
              "clef": "9/18",
              "clef-flash": "8/18",
              "kev-4b": "9/18",
              "lev": "6/18",
              "laya": "3/18",
              "julia-1": "4/18"
            },
            {
              "suite": "Games",
              "family": "sudoku-single",
              "jev": "6/17",
              "clef": "5/17",
              "clef-flash": "2/17",
              "kev-4b": "6/17",
              "lev": "3/17",
              "laya": "2/17",
              "julia-1": "2/17"
            },
            {
              "suite": "Games",
              "family": "wordle",
              "jev": "10/16",
              "clef": "7/16",
              "clef-flash": "3/16",
              "kev-4b": "9/16",
              "lev": "9/16",
              "laya": "3/16",
              "julia-1": "4/16"
            },
            {
              "suite": "Games",
              "family": "maze",
              "jev": "7/16",
              "clef": "4/16",
              "clef-flash": "2/16",
              "kev-4b": "5/16",
              "lev": "3/16",
              "laya": "5/16",
              "julia-1": "4/16"
            },
            {
              "suite": "Games",
              "family": "hanoi",
              "jev": "5/16",
              "clef": "6/16",
              "clef-flash": "8/16",
              "kev-4b": "4/16",
              "lev": "2/16",
              "laya": "4/16",
              "julia-1": "5/16"
            },
            {
              "suite": "Games",
              "family": "coin-weighing",
              "jev": "11/17",
              "clef": "11/17",
              "clef-flash": "9/17",
              "kev-4b": "9/17",
              "lev": "8/17",
              "laya": "4/17",
              "julia-1": "6/17"
            },
            {
              "suite": "Games",
              "family": "knight-moves",
              "jev": "9/21",
              "clef": "9/21",
              "clef-flash": "6/21",
              "kev-4b": "12/21",
              "lev": "4/21",
              "laya": "3/21",
              "julia-1": "3/21"
            },
            {
              "suite": "Logic and thought experiments",
              "family": "syllogism",
              "jev": "13/13",
              "clef": "12/13",
              "clef-flash": "12/13",
              "kev-4b": "13/13",
              "lev": "9/13",
              "laya": "7/13",
              "julia-1": "4/13"
            },
            {
              "suite": "Logic and thought experiments",
              "family": "quantifier-logic",
              "jev": "6/6",
              "clef": "5/6",
              "clef-flash": "2/6",
              "kev-4b": "4/6",
              "lev": "5/6",
              "laya": "1/6",
              "julia-1": "2/6"
            },
            {
              "suite": "Logic and thought experiments",
              "family": "conditional-logic",
              "jev": "12/12",
              "clef": "11/12",
              "clef-flash": "10/12",
              "kev-4b": "10/12",
              "lev": "7/12",
              "laya": "3/12",
              "julia-1": "7/12"
            },
            {
              "suite": "Logic and thought experiments",
              "family": "wason-selection",
              "jev": "5/6",
              "clef": "5/6",
              "clef-flash": "6/6",
              "kev-4b": "5/6",
              "lev": "3/6",
              "laya": "1/6",
              "julia-1": "1/6"
            },
            {
              "suite": "Logic and thought experiments",
              "family": "knights-knaves",
              "jev": "4/12",
              "clef": "2/12",
              "clef-flash": "2/12",
              "kev-4b": "2/12",
              "lev": "2/12",
              "laya": "3/12",
              "julia-1": "3/12"
            },
            {
              "suite": "Logic and thought experiments",
              "family": "seating-order",
              "jev": "7/12",
              "clef": "5/12",
              "clef-flash": "3/12",
              "kev-4b": "5/12",
              "lev": "3/12",
              "laya": "1/12",
              "julia-1": "4/12"
            },
            {
              "suite": "Logic and thought experiments",
              "family": "monty-hall",
              "jev": "7/10",
              "clef": "8/10",
              "clef-flash": "7/10",
              "kev-4b": "4/10",
              "lev": "9/10",
              "laya": "3/10",
              "julia-1": "3/10"
            },
            {
              "suite": "Logic and thought experiments",
              "family": "base-rate",
              "jev": "9/10",
              "clef": "9/10",
              "clef-flash": "3/10",
              "kev-4b": "2/10",
              "lev": "3/10",
              "laya": "2/10",
              "julia-1": "2/10"
            },
            {
              "suite": "Logic and thought experiments",
              "family": "dice-cards",
              "jev": "9/10",
              "clef": "6/10",
              "clef-flash": "6/10",
              "kev-4b": "5/10",
              "lev": "6/10",
              "laya": "4/10",
              "julia-1": "4/10"
            },
            {
              "suite": "Logic and thought experiments",
              "family": "gamble-ev",
              "jev": "6/9",
              "clef": "9/9",
              "clef-flash": "6/9",
              "kev-4b": "9/9",
              "lev": "5/9",
              "laya": "1/9",
              "julia-1": "2/9"
            },
            {
              "suite": "Logic and thought experiments",
              "family": "sunk-cost",
              "jev": "5/6",
              "clef": "5/6",
              "clef-flash": "4/6",
              "kev-4b": "5/6",
              "lev": "5/6",
              "laya": "2/6",
              "julia-1": "2/6"
            },
            {
              "suite": "Logic and thought experiments",
              "family": "gamblers-fallacy",
              "jev": "9/11",
              "clef": "9/11",
              "clef-flash": "7/11",
              "kev-4b": "7/11",
              "lev": "8/11",
              "laya": "4/11",
              "julia-1": "4/11"
            },
            {
              "suite": "Logic and thought experiments",
              "family": "denominator-neglect",
              "jev": "1/3",
              "clef": "1/3",
              "clef-flash": "1/3",
              "kev-4b": "1/3",
              "lev": "2/3",
              "laya": "1/3",
              "julia-1": "2/3"
            },
            {
              "suite": "Logic and thought experiments",
              "family": "conjunction",
              "jev": "4/5",
              "clef": "2/5",
              "clef-flash": "4/5",
              "kev-4b": "2/5",
              "lev": "2/5",
              "laya": "2/5",
              "julia-1": "2/5"
            },
            {
              "suite": "Logic and thought experiments",
              "family": "simpson",
              "jev": "2/4",
              "clef": "0/4",
              "clef-flash": "1/4",
              "kev-4b": "1/4",
              "lev": "2/4",
              "laya": "1/4",
              "julia-1": "1/4"
            },
            {
              "suite": "Logic and thought experiments",
              "family": "sample-size",
              "jev": "4/5",
              "clef": "4/5",
              "clef-flash": "5/5",
              "kev-4b": "1/5",
              "lev": "1/5",
              "laya": "1/5",
              "julia-1": "3/5"
            },
            {
              "suite": "Logic and thought experiments",
              "family": "anchoring",
              "jev": "4/4",
              "clef": "2/4",
              "clef-flash": "3/4",
              "kev-4b": "2/4",
              "lev": "2/4",
              "laya": "2/4",
              "julia-1": "1/4"
            },
            {
              "suite": "Logic and thought experiments",
              "family": "units",
              "jev": "4/9",
              "clef": "3/9",
              "clef-flash": "4/9",
              "kev-4b": "3/9",
              "lev": "2/9",
              "laya": "2/9",
              "julia-1": "5/9"
            },
            {
              "suite": "Logic and thought experiments",
              "family": "calendar-time",
              "jev": "5/9",
              "clef": "3/9",
              "clef-flash": "5/9",
              "kev-4b": "4/9",
              "lev": "3/9",
              "laya": "3/9",
              "julia-1": "2/9"
            },
            {
              "suite": "Policy cases, 20 industries",
              "family": "insurance",
              "jev": "10/17",
              "clef": "11/17",
              "clef-flash": "15/17",
              "kev-4b": "9/17",
              "lev": "8/17",
              "laya": "8/17",
              "julia-1": "4/17"
            },
            {
              "suite": "Policy cases, 20 industries",
              "family": "retail",
              "jev": "9/13",
              "clef": "7/13",
              "clef-flash": "10/13",
              "kev-4b": "7/13",
              "lev": "7/13",
              "laya": "2/13",
              "julia-1": "1/13"
            },
            {
              "suite": "Policy cases, 20 industries",
              "family": "banking",
              "jev": "12/16",
              "clef": "13/16",
              "clef-flash": "10/16",
              "kev-4b": "8/16",
              "lev": "11/16",
              "laya": "3/16",
              "julia-1": "1/16"
            },
            {
              "suite": "Policy cases, 20 industries",
              "family": "fraud",
              "jev": "12/14",
              "clef": "10/14",
              "clef-flash": "11/14",
              "kev-4b": "11/14",
              "lev": "11/14",
              "laya": "3/14",
              "julia-1": "5/14"
            },
            {
              "suite": "Policy cases, 20 industries",
              "family": "airline",
              "jev": "14/14",
              "clef": "11/14",
              "clef-flash": "13/14",
              "kev-4b": "10/14",
              "lev": "11/14",
              "laya": "2/14",
              "julia-1": "4/14"
            },
            {
              "suite": "Policy cases, 20 industries",
              "family": "hotel",
              "jev": "14/16",
              "clef": "11/16",
              "clef-flash": "10/16",
              "kev-4b": "8/16",
              "lev": "6/16",
              "laya": "4/16",
              "julia-1": "4/16"
            },
            {
              "suite": "Policy cases, 20 industries",
              "family": "logistics",
              "jev": "13/13",
              "clef": "10/13",
              "clef-flash": "10/13",
              "kev-4b": "9/13",
              "lev": "9/13",
              "laya": "1/13",
              "julia-1": "4/13"
            },
            {
              "suite": "Policy cases, 20 industries",
              "family": "saas",
              "jev": "13/13",
              "clef": "13/13",
              "clef-flash": "11/13",
              "kev-4b": "10/13",
              "lev": "11/13",
              "laya": "3/13",
              "julia-1": "2/13"
            },
            {
              "suite": "Policy cases, 20 industries",
              "family": "telecom",
              "jev": "15/15",
              "clef": "13/15",
              "clef-flash": "11/15",
              "kev-4b": "12/15",
              "lev": "9/15",
              "laya": "4/15",
              "julia-1": "3/15"
            },
            {
              "suite": "Policy cases, 20 industries",
              "family": "utilities",
              "jev": "8/14",
              "clef": "10/14",
              "clef-flash": "10/14",
              "kev-4b": "11/14",
              "lev": "9/14",
              "laya": "5/14",
              "julia-1": "3/14"
            },
            {
              "suite": "Policy cases, 20 industries",
              "family": "hr-leave",
              "jev": "13/13",
              "clef": "12/13",
              "clef-flash": "9/13",
              "kev-4b": "12/13",
              "lev": "10/13",
              "laya": "3/13",
              "julia-1": "3/13"
            },
            {
              "suite": "Policy cases, 20 industries",
              "family": "hr-expense",
              "jev": "10/12",
              "clef": "11/12",
              "clef-flash": "11/12",
              "kev-4b": "10/12",
              "lev": "9/12",
              "laya": "4/12",
              "julia-1": "5/12"
            },
            {
              "suite": "Policy cases, 20 industries",
              "family": "realestate",
              "jev": "9/15",
              "clef": "4/15",
              "clef-flash": "7/15",
              "kev-4b": "9/15",
              "lev": "2/15",
              "laya": "6/15",
              "julia-1": "5/15"
            },
            {
              "suite": "Policy cases, 20 industries",
              "family": "education",
              "jev": "13/15",
              "clef": "8/15",
              "clef-flash": "9/15",
              "kev-4b": "8/15",
              "lev": "6/15",
              "laya": "5/15",
              "julia-1": "6/15"
            },
            {
              "suite": "Policy cases, 20 industries",
              "family": "food",
              "jev": "10/15",
              "clef": "13/15",
              "clef-flash": "10/15",
              "kev-4b": "12/15",
              "lev": "11/15",
              "laya": "4/15",
              "julia-1": "5/15"
            },
            {
              "suite": "Policy cases, 20 industries",
              "family": "manufacturing",
              "jev": "9/12",
              "clef": "6/12",
              "clef-flash": "9/12",
              "kev-4b": "5/12",
              "lev": "4/12",
              "laya": "3/12",
              "julia-1": "5/12"
            },
            {
              "suite": "Policy cases, 20 industries",
              "family": "health-billing",
              "jev": "10/12",
              "clef": "10/12",
              "clef-flash": "8/12",
              "kev-4b": "7/12",
              "lev": "6/12",
              "laya": "3/12",
              "julia-1": "2/12"
            },
            {
              "suite": "Policy cases, 20 industries",
              "family": "health-scheduling",
              "jev": "3/7",
              "clef": "3/7",
              "clef-flash": "3/7",
              "kev-4b": "5/7",
              "lev": "5/7",
              "laya": "2/7",
              "julia-1": "0/7"
            },
            {
              "suite": "Policy cases, 20 industries",
              "family": "permits",
              "jev": "14/14",
              "clef": "12/14",
              "clef-flash": "12/14",
              "kev-4b": "11/14",
              "lev": "9/14",
              "laya": "2/14",
              "julia-1": "2/14"
            },
            {
              "suite": "Policy cases, 20 industries",
              "family": "itsec",
              "jev": "12/14",
              "clef": "9/14",
              "clef-flash": "9/14",
              "kev-4b": "11/14",
              "lev": "8/14",
              "laya": "6/14",
              "julia-1": "5/14"
            },
            {
              "suite": "Usability intents",
              "family": "clear intent",
              "jev": "28/31",
              "clef": "29/31",
              "clef-flash": "31/31",
              "kev-4b": "27/31",
              "lev": "30/31",
              "laya": "16/31",
              "julia-1": "8/31"
            },
            {
              "suite": "Usability intents",
              "family": "screen or history context",
              "jev": "38/39",
              "clef": "33/39",
              "clef-flash": "34/39",
              "kev-4b": "29/39",
              "lev": "33/39",
              "laya": "7/39",
              "julia-1": "3/39"
            },
            {
              "suite": "Usability intents",
              "family": "out of scope",
              "jev": "18/19",
              "clef": "18/19",
              "clef-flash": "17/19",
              "kev-4b": "17/19",
              "lev": "14/19",
              "laya": "12/19",
              "julia-1": "0/19"
            },
            {
              "suite": "Usability intents",
              "family": "ambiguous request",
              "jev": "17/20",
              "clef": "13/20",
              "clef-flash": "7/20",
              "kev-4b": "3/20",
              "lev": "6/20",
              "laya": "1/20",
              "julia-1": "8/20"
            },
            {
              "suite": "Usability intents",
              "family": "destructive needs confirm",
              "jev": "27/27",
              "clef": "24/27",
              "clef-flash": "25/27",
              "kev-4b": "23/27",
              "lev": "25/27",
              "laya": "5/27",
              "julia-1": "4/27"
            },
            {
              "suite": "Usability intents",
              "family": "slang and typos",
              "jev": "17/18",
              "clef": "18/18",
              "clef-flash": "17/18",
              "kev-4b": "16/18",
              "lev": "17/18",
              "laya": "5/18",
              "julia-1": "3/18"
            },
            {
              "suite": "Usability intents",
              "family": "multilingual",
              "jev": "33/34",
              "clef": "29/34",
              "clef-flash": "28/34",
              "kev-4b": "31/34",
              "lev": "29/34",
              "laya": "13/34",
              "julia-1": "6/34"
            },
            {
              "suite": "Usability intents",
              "family": "accessibility",
              "jev": "13/17",
              "clef": "16/17",
              "clef-flash": "17/17",
              "kev-4b": "15/17",
              "lev": "17/17",
              "laya": "5/17",
              "julia-1": "1/17"
            },
            {
              "suite": "Usability intents",
              "family": "undo, redo and back",
              "jev": "20/21",
              "clef": "21/21",
              "clef-flash": "20/21",
              "kev-4b": "20/21",
              "lev": "15/21",
              "laya": "9/21",
              "julia-1": "3/21"
            },
            {
              "suite": "Stress tests",
              "family": "injection",
              "jev": "38/40",
              "clef": "36/40",
              "clef-flash": "31/40",
              "kev-4b": "38/40",
              "lev": "27/40",
              "laya": "13/40",
              "julia-1": "18/40"
            },
            {
              "suite": "Stress tests",
              "family": "needle",
              "jev": "22/22",
              "clef": "21/22",
              "clef-flash": "15/22",
              "kev-4b": "16/22",
              "lev": "15/22",
              "laya": "7/22",
              "julia-1": "8/22"
            },
            {
              "suite": "Stress tests",
              "family": "distractor",
              "jev": "15/16",
              "clef": "15/16",
              "clef-flash": "13/16",
              "kev-4b": "12/16",
              "lev": "9/16",
              "laya": "3/16",
              "julia-1": "4/16"
            },
            {
              "suite": "Stress tests",
              "family": "negation",
              "jev": "17/21",
              "clef": "19/21",
              "clef-flash": "18/21",
              "kev-4b": "13/21",
              "lev": "16/21",
              "laya": "8/21",
              "julia-1": "6/21"
            },
            {
              "suite": "Stress tests",
              "family": "threshold",
              "jev": "13/25",
              "clef": "13/25",
              "clef-flash": "7/25",
              "kev-4b": "7/25",
              "lev": "12/25",
              "laya": "12/25",
              "julia-1": "12/25"
            },
            {
              "suite": "Stress tests",
              "family": "out-of-scope",
              "jev": "19/19",
              "clef": "18/19",
              "clef-flash": "16/19",
              "kev-4b": "16/19",
              "lev": "15/19",
              "laya": "4/19",
              "julia-1": "9/19"
            },
            {
              "suite": "Stress tests",
              "family": "near-options",
              "jev": "12/14",
              "clef": "9/14",
              "clef-flash": "10/14",
              "kev-4b": "12/14",
              "lev": "10/14",
              "laya": "4/14",
              "julia-1": "4/14"
            },
            {
              "suite": "Stress tests",
              "family": "many-options",
              "jev": "16/16",
              "clef": "14/16",
              "clef-flash": "9/16",
              "kev-4b": "14/16",
              "lev": "11/16",
              "laya": "1/16",
              "julia-1": "0/16"
            },
            {
              "suite": "Stress tests",
              "family": "paraphrase",
              "jev": "22/27",
              "clef": "20/27",
              "clef-flash": "20/27",
              "kev-4b": "18/27",
              "lev": "14/27",
              "laya": "11/27",
              "julia-1": "11/27"
            }
          ]
        },
        {
          "id": "arena-industries",
          "title": "Policy cases by industry: right answers per model (first presentation)",
          "columns": [
            {
              "key": "industry",
              "label": "Industry",
              "unit": "text"
            },
            {
              "key": "jev",
              "label": "Jev 1.13",
              "unit": "text"
            },
            {
              "key": "clef",
              "label": "Clef 27B",
              "unit": "text"
            },
            {
              "key": "clef-flash",
              "label": "Clef-Flash 9B",
              "unit": "text"
            },
            {
              "key": "kev-4b",
              "label": "Kev 4B",
              "unit": "text"
            },
            {
              "key": "lev",
              "label": "lev 4B",
              "unit": "text"
            },
            {
              "key": "laya",
              "label": "Laya",
              "unit": "text"
            },
            {
              "key": "julia-1",
              "label": "Julia-1",
              "unit": "text"
            }
          ],
          "rows": [
            {
              "industry": "Airline rebooking",
              "jev": "14/14",
              "clef": "11/14",
              "clef-flash": "13/14",
              "kev-4b": "10/14",
              "lev": "11/14",
              "laya": "2/14",
              "julia-1": "4/14"
            },
            {
              "industry": "Banking and card disputes",
              "jev": "12/16",
              "clef": "13/16",
              "clef-flash": "10/16",
              "kev-4b": "8/16",
              "lev": "11/16",
              "laya": "3/16",
              "julia-1": "1/16"
            },
            {
              "industry": "Education admissions and registration",
              "jev": "13/15",
              "clef": "8/15",
              "clef-flash": "9/15",
              "kev-4b": "8/15",
              "lev": "6/15",
              "laya": "5/15",
              "julia-1": "6/15"
            },
            {
              "industry": "HR leave and expense policy",
              "jev": "23/25",
              "clef": "23/25",
              "clef-flash": "20/25",
              "kev-4b": "22/25",
              "lev": "19/25",
              "laya": "7/25",
              "julia-1": "8/25"
            },
            {
              "industry": "Healthcare administration",
              "jev": "13/19",
              "clef": "13/19",
              "clef-flash": "11/19",
              "kev-4b": "12/19",
              "lev": "11/19",
              "laya": "5/19",
              "julia-1": "2/19"
            },
            {
              "industry": "Hotel booking changes",
              "jev": "14/16",
              "clef": "11/16",
              "clef-flash": "10/16",
              "kev-4b": "8/16",
              "lev": "6/16",
              "laya": "4/16",
              "julia-1": "4/16"
            },
            {
              "industry": "IT security access requests",
              "jev": "12/14",
              "clef": "9/14",
              "clef-flash": "9/14",
              "kev-4b": "11/14",
              "lev": "8/14",
              "laya": "6/14",
              "julia-1": "5/14"
            },
            {
              "industry": "Insurance claims",
              "jev": "10/17",
              "clef": "11/17",
              "clef-flash": "15/17",
              "kev-4b": "9/17",
              "lev": "8/17",
              "laya": "8/17",
              "julia-1": "4/17"
            },
            {
              "industry": "Logistics and shipping exceptions",
              "jev": "13/13",
              "clef": "10/13",
              "clef-flash": "10/13",
              "kev-4b": "9/13",
              "lev": "9/13",
              "laya": "1/13",
              "julia-1": "4/13"
            },
            {
              "industry": "Manufacturing quality control",
              "jev": "9/12",
              "clef": "6/12",
              "clef-flash": "9/12",
              "kev-4b": "5/12",
              "lev": "4/12",
              "laya": "3/12",
              "julia-1": "5/12"
            },
            {
              "industry": "Payments fraud flags",
              "jev": "12/14",
              "clef": "10/14",
              "clef-flash": "11/14",
              "kev-4b": "11/14",
              "lev": "11/14",
              "laya": "3/14",
              "julia-1": "5/14"
            },
            {
              "industry": "Public-sector permits",
              "jev": "14/14",
              "clef": "12/14",
              "clef-flash": "12/14",
              "kev-4b": "11/14",
              "lev": "9/14",
              "laya": "2/14",
              "julia-1": "2/14"
            },
            {
              "industry": "Real-estate rental applications",
              "jev": "9/15",
              "clef": "4/15",
              "clef-flash": "7/15",
              "kev-4b": "9/15",
              "lev": "2/15",
              "laya": "6/15",
              "julia-1": "5/15"
            },
            {
              "industry": "Restaurant and food delivery",
              "jev": "10/15",
              "clef": "13/15",
              "clef-flash": "10/15",
              "kev-4b": "12/15",
              "lev": "11/15",
              "laya": "4/15",
              "julia-1": "5/15"
            },
            {
              "industry": "Retail returns",
              "jev": "9/13",
              "clef": "7/13",
              "clef-flash": "10/13",
              "kev-4b": "7/13",
              "lev": "7/13",
              "laya": "2/13",
              "julia-1": "1/13"
            },
            {
              "industry": "SaaS customer support routing",
              "jev": "13/13",
              "clef": "13/13",
              "clef-flash": "11/13",
              "kev-4b": "10/13",
              "lev": "11/13",
              "laya": "3/13",
              "julia-1": "2/13"
            },
            {
              "industry": "Telecom plan changes",
              "jev": "15/15",
              "clef": "13/15",
              "clef-flash": "11/15",
              "kev-4b": "12/15",
              "lev": "9/15",
              "laya": "4/15",
              "julia-1": "3/15"
            },
            {
              "industry": "Utilities and energy billing",
              "jev": "8/14",
              "clef": "10/14",
              "clef-flash": "10/14",
              "kev-4b": "11/14",
              "lev": "9/14",
              "laya": "5/14",
              "julia-1": "3/14"
            }
          ]
        },
        {
          "id": "arena-games-table",
          "title": "Round robin by game: wins-draws-losses",
          "columns": [
            {
              "key": "model",
              "label": "Model",
              "unit": "text"
            },
            {
              "key": "ttt",
              "label": "Tic-tac-toe",
              "unit": "text"
            },
            {
              "key": "c4",
              "label": "Connect Four",
              "unit": "text"
            },
            {
              "key": "nim",
              "label": "Nim",
              "unit": "text"
            },
            {
              "key": "dab",
              "label": "Dots and Boxes",
              "unit": "text"
            },
            {
              "key": "illegal",
              "label": "Illegal moves",
              "unit": "count"
            }
          ],
          "rows": [
            {
              "model": "Jev 1.13",
              "ttt": "37-6-37",
              "c4": "56-0-24",
              "nim": "31-0-49",
              "dab": "59-0-21",
              "illegal": 0
            },
            {
              "model": "Clef 27B",
              "ttt": "38-5-37",
              "c4": "42-0-38",
              "nim": "45-0-35",
              "dab": "23-0-57",
              "illegal": 0
            },
            {
              "model": "Clef-Flash 9B",
              "ttt": "34-7-39",
              "c4": "38-0-42",
              "nim": "36-0-44",
              "dab": "31-0-49",
              "illegal": 0
            },
            {
              "model": "lev 4B",
              "ttt": "29-9-42",
              "c4": "45-0-35",
              "nim": "29-0-51",
              "dab": "34-0-46",
              "illegal": 0
            },
            {
              "model": "Kev 4B",
              "ttt": "33-9-38",
              "c4": "27-0-53",
              "nim": "38-0-42",
              "dab": "34-0-46",
              "illegal": 0
            },
            {
              "model": "Laya",
              "ttt": "20-13-47",
              "c4": "17-0-63",
              "nim": "34-0-46",
              "dab": "44-0-36",
              "illegal": 0
            },
            {
              "model": "Julia-1",
              "ttt": "29-6-45",
              "c4": "26-0-54",
              "nim": "38-0-42",
              "dab": "23-0-57",
              "illegal": 0
            }
          ]
        },
        {
          "id": "arena-pong-table",
          "title": "Pong round robin: every model, speed and accuracy side by side",
          "columns": [
            {
              "key": "model",
              "label": "Model",
              "unit": "text"
            },
            {
              "key": "record",
              "label": "Games W-D-L",
              "unit": "text"
            },
            {
              "key": "medianMs",
              "label": "Median decision (ms)",
              "unit": "ms"
            },
            {
              "key": "decisions",
              "label": "Decisions",
              "unit": "count"
            },
            {
              "key": "perfect",
              "label": "Right zone when the ball came",
              "unit": "text"
            },
            {
              "key": "returnRate",
              "label": "Shots returned",
              "unit": "text"
            }
          ],
          "rows": [
            {
              "model": "Jev 1.13",
              "record": "13-0-3",
              "medianMs": 132.4,
              "decisions": 1436,
              "perfect": "100% (523/523)",
              "returnRate": "86%"
            },
            {
              "model": "Kev 4B",
              "record": "12-0-4",
              "medianMs": 284.7,
              "decisions": 815,
              "perfect": "98% (303/310)",
              "returnRate": "80%"
            },
            {
              "model": "lev 4B",
              "record": "9-0-7",
              "medianMs": 417.8,
              "decisions": 451,
              "perfect": "99% (151/153)",
              "returnRate": "59%"
            },
            {
              "model": "Clef-Flash 9B",
              "record": "8-0-8",
              "medianMs": 527.6,
              "decisions": 336,
              "perfect": "83% (112/135)",
              "returnRate": "49%"
            },
            {
              "model": "Clef 27B",
              "record": "4-0-12",
              "medianMs": 1789.1,
              "decisions": 84,
              "perfect": "100% (31/31)",
              "returnRate": "26%"
            },
            {
              "model": "Julia-1",
              "record": "4-0-12",
              "medianMs": 12.4,
              "decisions": 7451,
              "perfect": "21% (603/2811)",
              "returnRate": "21%"
            },
            {
              "model": "Laya",
              "record": "3-0-13",
              "medianMs": 36.2,
              "decisions": 2460,
              "perfect": "12% (122/999)",
              "returnRate": "12%"
            }
          ]
        },
        {
          "id": "arena-pong-featured",
          "title": "Featured Pong series (4 games each, seed 2)",
          "columns": [
            {
              "key": "pair",
              "label": "Pair",
              "unit": "text"
            },
            {
              "key": "aWins",
              "label": "First won",
              "unit": "count"
            },
            {
              "key": "bWins",
              "label": "Second won",
              "unit": "count"
            }
          ],
          "rows": [
            {
              "pair": "Clef 27B vs Clef-Flash 9B",
              "aWins": 1,
              "bWins": 1
            },
            {
              "pair": "Clef 27B vs Jev 1.13",
              "aWins": 0,
              "bWins": 2
            },
            {
              "pair": "Jev 1.13 vs Laya",
              "aWins": 4,
              "bWins": 0
            }
          ]
        },
        {
          "id": "arena-pong-sides",
          "title": "Featured Pong match (Jev 1.13 vs Laya): who played which side",
          "columns": [
            {
              "key": "side",
              "label": "Side",
              "unit": "text"
            },
            {
              "key": "model",
              "label": "Model",
              "unit": "text"
            },
            {
              "key": "id",
              "label": "Id",
              "unit": "text"
            },
            {
              "key": "medianMs",
              "label": "Median decision time (ms)",
              "unit": "ms"
            },
            {
              "key": "decisions",
              "label": "Decisions in the match",
              "unit": "count"
            },
            {
              "key": "points",
              "label": "Points",
              "unit": "count"
            }
          ],
          "rows": [
            {
              "side": "left",
              "model": "Jev 1.13",
              "id": "jev",
              "medianMs": 123.6,
              "decisions": 54,
              "points": 3
            },
            {
              "side": "right",
              "model": "Laya",
              "id": "laya",
              "medianMs": 37.2,
              "decisions": 156,
              "points": 0
            }
          ]
        },
        {
          "id": "arena-c4-players",
          "title": "Featured Connect Four game: players and result",
          "columns": [
            {
              "key": "side",
              "label": "Side",
              "unit": "text"
            },
            {
              "key": "model",
              "label": "Model",
              "unit": "text"
            },
            {
              "key": "result",
              "label": "Result",
              "unit": "text"
            }
          ],
          "rows": [
            {
              "side": "A",
              "model": "Jev 1.13",
              "result": "won"
            },
            {
              "side": "B",
              "model": "Clef 27B",
              "result": "lost"
            }
          ]
        },
        {
          "id": "arena-pong-rally",
          "title": "Featured Pong rally (Jev 1.13 vs Laya): ball and paddle positions every 80 ms, 0 to 1",
          "columns": [
            {
              "key": "t",
              "label": "Seconds",
              "unit": "seconds"
            },
            {
              "key": "x",
              "label": "Ball x",
              "unit": "ratio"
            },
            {
              "key": "y",
              "label": "Ball y",
              "unit": "ratio"
            },
            {
              "key": "left",
              "label": "Left paddle",
              "unit": "ratio"
            },
            {
              "key": "right",
              "label": "Right paddle",
              "unit": "ratio"
            }
          ],
          "rows": [
            {
              "t": 0,
              "x": 0.5,
              "y": 0.5,
              "left": 0.5,
              "right": 0.5
            },
            {
              "t": 0.083,
              "x": 0.5,
              "y": 0.5,
              "left": 0.5,
              "right": 0.5
            },
            {
              "t": 0.167,
              "x": 0.5,
              "y": 0.5,
              "left": 0.5,
              "right": 0.5
            },
            {
              "t": 0.25,
              "x": 0.5,
              "y": 0.5,
              "left": 0.5,
              "right": 0.5
            },
            {
              "t": 0.333,
              "x": 0.5,
              "y": 0.5,
              "left": 0.5,
              "right": 0.5
            },
            {
              "t": 0.417,
              "x": 0.5,
              "y": 0.5,
              "left": 0.5,
              "right": 0.5
            },
            {
              "t": 0.5,
              "x": 0.5,
              "y": 0.5,
              "left": 0.5,
              "right": 0.5
            },
            {
              "t": 0.583,
              "x": 0.5488,
              "y": 0.471,
              "left": 0.5,
              "right": 0.48
            },
            {
              "t": 0.667,
              "x": 0.5975,
              "y": 0.441,
              "left": 0.5,
              "right": 0.43
            },
            {
              "t": 0.75,
              "x": 0.6463,
              "y": 0.412,
              "left": 0.49,
              "right": 0.38
            },
            {
              "t": 0.833,
              "x": 0.695,
              "y": 0.382,
              "left": 0.44,
              "right": 0.33
            },
            {
              "t": 0.917,
              "x": 0.7438,
              "y": 0.353,
              "left": 0.39,
              "right": 0.3
            },
            {
              "t": 1,
              "x": 0.7925,
              "y": 0.323,
              "left": 0.34,
              "right": 0.3
            },
            {
              "t": 1.083,
              "x": 0.8413,
              "y": 0.294,
              "left": 0.29,
              "right": 0.3
            },
            {
              "t": 1.167,
              "x": 0.89,
              "y": 0.264,
              "left": 0.24,
              "right": 0.3
            },
            {
              "t": 1.25,
              "x": 0.9106,
              "y": 0.235,
              "left": 0.19,
              "right": 0.28
            },
            {
              "t": 1.333,
              "x": 0.8563,
              "y": 0.206,
              "left": 0.14,
              "right": 0.25
            },
            {
              "t": 1.417,
              "x": 0.8019,
              "y": 0.177,
              "left": 0.13,
              "right": 0.3
            },
            {
              "t": 1.5,
              "x": 0.7475,
              "y": 0.149,
              "left": 0.18,
              "right": 0.3
            },
            {
              "t": 1.583,
              "x": 0.6931,
              "y": 0.12,
              "left": 0.23,
              "right": 0.3
            },
            {
              "t": 1.667,
              "x": 0.6388,
              "y": 0.091,
              "left": 0.28,
              "right": 0.3
            },
            {
              "t": 1.75,
              "x": 0.5838,
              "y": 0.062,
              "left": 0.3,
              "right": 0.29
            },
            {
              "t": 1.833,
              "x": 0.5294,
              "y": 0.033,
              "left": 0.3,
              "right": 0.3
            },
            {
              "t": 1.917,
              "x": 0.475,
              "y": 0.035,
              "left": 0.3,
              "right": 0.3
            },
            {
              "t": 2,
              "x": 0.4206,
              "y": 0.064,
              "left": 0.3,
              "right": 0.3
            },
            {
              "t": 2.083,
              "x": 0.3663,
              "y": 0.093,
              "left": 0.3,
              "right": 0.3
            },
            {
              "t": 2.167,
              "x": 0.3119,
              "y": 0.122,
              "left": 0.3,
              "right": 0.3
            },
            {
              "t": 2.25,
              "x": 0.2575,
              "y": 0.151,
              "left": 0.3,
              "right": 0.3
            },
            {
              "t": 2.333,
              "x": 0.2031,
              "y": 0.179,
              "left": 0.3,
              "right": 0.3
            },
            {
              "t": 2.417,
              "x": 0.1488,
              "y": 0.208,
              "left": 0.3,
              "right": 0.3
            },
            {
              "t": 2.5,
              "x": 0.0944,
              "y": 0.237,
              "left": 0.3,
              "right": 0.3
            },
            {
              "t": 2.583,
              "x": 0.1125,
              "y": 0.227,
              "left": 0.3,
              "right": 0.3
            },
            {
              "t": 2.667,
              "x": 0.1719,
              "y": 0.192,
              "left": 0.3,
              "right": 0.27
            },
            {
              "t": 2.75,
              "x": 0.2306,
              "y": 0.156,
              "left": 0.3,
              "right": 0.26
            },
            {
              "t": 2.833,
              "x": 0.2894,
              "y": 0.121,
              "left": 0.3,
              "right": 0.23
            },
            {
              "t": 2.917,
              "x": 0.3488,
              "y": 0.085,
              "left": 0.28,
              "right": 0.18
            },
            {
              "t": 3,
              "x": 0.4075,
              "y": 0.05,
              "left": 0.23,
              "right": 0.13
            },
            {
              "t": 3.083,
              "x": 0.4669,
              "y": 0.026,
              "left": 0.18,
              "right": 0.1
            },
            {
              "t": 3.167,
              "x": 0.5256,
              "y": 0.061,
              "left": 0.13,
              "right": 0.1
            },
            {
              "t": 3.25,
              "x": 0.5844,
              "y": 0.097,
              "left": 0.1,
              "right": 0.1
            },
            {
              "t": 3.333,
              "x": 0.6438,
              "y": 0.132,
              "left": 0.1,
              "right": 0.1
            },
            {
              "t": 3.417,
              "x": 0.7025,
              "y": 0.167,
              "left": 0.1,
              "right": 0.1
            },
            {
              "t": 3.5,
              "x": 0.7619,
              "y": 0.203,
              "left": 0.1,
              "right": 0.1
            },
            {
              "t": 3.583,
              "x": 0.8206,
              "y": 0.238,
              "left": 0.1,
              "right": 0.1
            },
            {
              "t": 3.667,
              "x": 0.8794,
              "y": 0.274,
              "left": 0.1,
              "right": 0.1
            },
            {
              "t": 3.75,
              "x": 0.9388,
              "y": 0.309,
              "left": 0.1,
              "right": 0.1
            },
            {
              "t": 3.833,
              "x": 0.9975,
              "y": 0.345,
              "left": 0.15,
              "right": 0.11
            }
          ]
        },
        {
          "id": "arena-c4-game",
          "title": "Featured Connect Four game (Jev 1.13 vs Clef 27B): every move. Rule: The longest Connect Four game between jev and clef in the turn-based round robin (most moves, openings included); ties go to the earliest.",
          "columns": [
            {
              "key": "moveNo",
              "label": "Move",
              "unit": "count"
            },
            {
              "key": "opening",
              "label": "Random opening move",
              "unit": "text"
            },
            {
              "key": "side",
              "label": "Side",
              "unit": "text"
            },
            {
              "key": "col",
              "label": "Column",
              "unit": "count"
            },
            {
              "key": "p",
              "label": "Probability",
              "unit": "ratio"
            },
            {
              "key": "optimal",
              "label": "Perfect move",
              "unit": "text"
            },
            {
              "key": "best",
              "label": "Best columns",
              "unit": "text"
            }
          ],
          "rows": [
            {
              "moveNo": 1,
              "opening": true,
              "side": "A",
              "col": 4,
              "p": null,
              "optimal": null,
              "best": ""
            },
            {
              "moveNo": 2,
              "opening": true,
              "side": "B",
              "col": 4,
              "p": null,
              "optimal": null,
              "best": ""
            },
            {
              "moveNo": 3,
              "opening": true,
              "side": "A",
              "col": 3,
              "p": null,
              "optimal": null,
              "best": ""
            },
            {
              "moveNo": 4,
              "opening": true,
              "side": "B",
              "col": 3,
              "p": null,
              "optimal": null,
              "best": ""
            },
            {
              "moveNo": 5,
              "opening": false,
              "side": "A",
              "col": 4,
              "p": 0.44,
              "optimal": false,
              "best": "2,5"
            },
            {
              "moveNo": 6,
              "opening": false,
              "side": "B",
              "col": 4,
              "p": 0.1926,
              "optimal": false,
              "best": "5"
            },
            {
              "moveNo": 7,
              "opening": false,
              "side": "A",
              "col": 3,
              "p": 0.39,
              "optimal": false,
              "best": "2,5"
            },
            {
              "moveNo": 8,
              "opening": false,
              "side": "B",
              "col": 3,
              "p": 0.1899,
              "optimal": false,
              "best": "5"
            },
            {
              "moveNo": 9,
              "opening": false,
              "side": "A",
              "col": 3,
              "p": 0.44,
              "optimal": false,
              "best": "2,5"
            },
            {
              "moveNo": 10,
              "opening": false,
              "side": "B",
              "col": 4,
              "p": 0.1815,
              "optimal": false,
              "best": "2"
            },
            {
              "moveNo": 11,
              "opening": false,
              "side": "A",
              "col": 3,
              "p": 0.58,
              "optimal": false,
              "best": "2,5"
            },
            {
              "moveNo": 12,
              "opening": false,
              "side": "B",
              "col": 4,
              "p": 0.2549,
              "optimal": false,
              "best": "2"
            },
            {
              "moveNo": 13,
              "opening": false,
              "side": "A",
              "col": 5,
              "p": 0.62,
              "optimal": true,
              "best": "2,5"
            },
            {
              "moveNo": 14,
              "opening": false,
              "side": "B",
              "col": 2,
              "p": 0.2223,
              "optimal": null,
              "best": ""
            },
            {
              "moveNo": 15,
              "opening": false,
              "side": "A",
              "col": 5,
              "p": 0.6,
              "optimal": false,
              "best": "6"
            },
            {
              "moveNo": 16,
              "opening": false,
              "side": "B",
              "col": 6,
              "p": 0.2136,
              "optimal": true,
              "best": "6"
            },
            {
              "moveNo": 17,
              "opening": false,
              "side": "A",
              "col": 5,
              "p": 0.61,
              "optimal": true,
              "best": "5"
            },
            {
              "moveNo": 18,
              "opening": false,
              "side": "B",
              "col": 5,
              "p": 0.2408,
              "optimal": true,
              "best": "5"
            },
            {
              "moveNo": 19,
              "opening": false,
              "side": "A",
              "col": 5,
              "p": 0.67,
              "optimal": false,
              "best": "2"
            },
            {
              "moveNo": 20,
              "opening": false,
              "side": "B",
              "col": 5,
              "p": 0.2192,
              "optimal": true,
              "best": "5"
            },
            {
              "moveNo": 21,
              "opening": false,
              "side": "A",
              "col": 6,
              "p": 0.48,
              "optimal": true,
              "best": "6"
            },
            {
              "moveNo": 22,
              "opening": false,
              "side": "B",
              "col": 6,
              "p": 0.3235,
              "optimal": true,
              "best": "6"
            },
            {
              "moveNo": 23,
              "opening": false,
              "side": "A",
              "col": 6,
              "p": 0.74,
              "optimal": true,
              "best": "6"
            },
            {
              "moveNo": 24,
              "opening": false,
              "side": "B",
              "col": 6,
              "p": 0.3626,
              "optimal": false,
              "best": "1"
            },
            {
              "moveNo": 25,
              "opening": false,
              "side": "A",
              "col": 6,
              "p": 0.62,
              "optimal": true,
              "best": "6"
            },
            {
              "moveNo": 26,
              "opening": false,
              "side": "B",
              "col": 2,
              "p": 0.4789,
              "optimal": false,
              "best": "7"
            },
            {
              "moveNo": 27,
              "opening": false,
              "side": "A",
              "col": 2,
              "p": 0.54,
              "optimal": true,
              "best": "2"
            }
          ]
        },
        {
          "id": "arena-leagues-table",
          "title": "Arena leagues: wins-draws-losses and Elo per model",
          "columns": [
            {
              "key": "model",
              "label": "Player",
              "unit": "text"
            },
            {
              "key": "c0",
              "label": "Othello",
              "unit": "text"
            },
            {
              "key": "c1",
              "label": "Pong, fair mode 150 ms",
              "unit": "text"
            },
            {
              "key": "c2",
              "label": "Snake, fair mode 150 ms",
              "unit": "text"
            },
            {
              "key": "c3",
              "label": "Tron, fair mode 150 ms",
              "unit": "text"
            }
          ],
          "rows": [
            {
              "model": "perfect",
              "c0": "32-0-0 (1280)",
              "c1": "7-0-0 (1098)",
              "c2": "–",
              "c3": "–"
            },
            {
              "model": "Clef 27B",
              "c0": "20-2-10 (1082)",
              "c1": "–",
              "c2": "11-0-5 (1068)",
              "c3": "12-3-17 (956)"
            },
            {
              "model": "random",
              "c0": "17-0-15 (1015)",
              "c1": "1-0-6 (930)",
              "c2": "6-0-10 (954)",
              "c3": "8-4-20 (898)"
            },
            {
              "model": "Laya",
              "c0": "15-1-16 (990)",
              "c1": "0-0-7 (902)",
              "c2": "1-1-14 (850)",
              "c3": "7-1-24 (852)"
            },
            {
              "model": "Julia-1",
              "c0": "13-2-17 (965)",
              "c1": "2-0-5 (958)",
              "c2": "9-0-7 (1022)",
              "c3": "2-1-29 (761)"
            },
            {
              "model": "lev 4B",
              "c0": "14-0-18 (964)",
              "c1": "4-0-3 (1014)",
              "c2": "4-0-12 (908)",
              "c3": "14-3-15 (991)"
            },
            {
              "model": "Clef-Flash 9B",
              "c0": "11-1-20 (924)",
              "c1": "3-0-4 (986)",
              "c2": "9-1-6 (1034)",
              "c3": "17-1-14 (1025)"
            },
            {
              "model": "Kev 4B",
              "c0": "9-1-22 (891)",
              "c1": "6-0-1 (1070)",
              "c2": "2-1-13 (875)",
              "c3": "17-2-13 (1031)"
            },
            {
              "model": "Jev 1.13",
              "c0": "9-1-22 (890)",
              "c1": "5-0-2 (1042)",
              "c2": "12-1-3 (1103)",
              "c3": "27-3-2 (1219)"
            },
            {
              "model": "expert",
              "c0": "–",
              "c1": "–",
              "c2": "16-0-0 (1185)",
              "c3": "30-2-0 (1267)"
            }
          ]
        }
      ],
      "related": [
        "routing-jev-vs-llm",
        "routing-overhead"
      ]
    },
    {
      "slug": "routing-jev-vs-llm",
      "title": "Jev vs Claude as a router: accuracy and cost",
      "seoTitle": "Jev router vs Claude Haiku and Sonnet: routing and cost",
      "description": "Typed routing decisions: Jev, a dedicated router model, against Claude Haiku and Sonnet. Accuracy with intervals, cost per 1,000 decisions, latency.",
      "question": "Should a small dedicated router or a general LLM make the platform’s typed routing decisions?",
      "answer": "Exact decisions: Jev 1.13 (TypeSafe) 221 of 246 live calls (90%; 74, 73 and 74 of 82 per repeat; case-level interval 82% to 95%); Claude Haiku 4.5 73 of 82 (89%, 80% to 94%); Claude Sonnet 5.5 77 of 82 (94%, 87% to 97%). The intervals overlap, so accuracy does not separate the routers here. Cost does: Jev costs $0.0337 per 1,000 decisions (a calculation from its reported input tokens) against Claude Haiku 4.5 $8.92, Claude Sonnet 5.5 $5.00. Case by case against Jev’s recorded production run (one pass), the exact McNemar test finds no difference (Claude Haiku 4.5 p = 1, Claude Sonnet 5.5 p = 0.375). Per decision, Jev is about 265x cheaper than Claude Haiku 4.5 and about 148x cheaper than Claude Sonnet 5.5. Median model time per decision through the Claude Code CLI: Claude Haiku 4.5 10.7 s, Claude Sonnet 5.5 1.6 s. Jev, called directly over HTTPS from one Mac, took a median 137 ms per call (p95 196 ms, 246 calls, wall time with the network inside it). That is a different route from the CLI, so it is not a model-against-model compute comparison. The case sets were tuned against Jev answers, which gives Jev a home advantage. Thought experiment (a calculation on 2,362 recorded calls, not a run): the same tokens cost $108.54 all on Sonnet 5.5, $161.62 under the platform policy mix (1.49x) and $105.53 with Haiku on side jobs only (2.8% less), because the main coding stages hold most of the spend.",
      "date": "2026-10-05",
      "updated": "2026-10-06",
      "tags": [
        "routing",
        "jev",
        "claude-haiku",
        "claude-sonnet",
        "model-routing",
        "thought-experiment"
      ],
      "method": [
        "Cases: the labelled decision suites the platform uses (failure class, message intent, is-it-a-rule, context shape), only the cases production asks a router about.",
        "Scoring: the repository’s own decision-eval runner. Exact means every scored question in a case was acceptable; key accuracy counts each question.",
        "Jev numbers come from a live run on 2026-10-06: 3 repeats of the same 82 decisions (246 counted calls), sent one at a time over HTTPS to the TypeSafe API from one Apple M3 Ultra Mac on a home network. A failed call would count as wrong; there were 0. Latency is client wall time, so the network is inside it, and the API reports no server time. Jev’s earlier recorded production run (2026-10-05, same case versions and runner) scored 74 of 82 and has no per-call latency. Cost is a calculation: reported input tokens × the published price.",
        "Claude routers ran through the Claude Code CLI with the production system text and schema, one call per decision, one pass over the 82 decisions.",
        "Stability: 73 of 82 decisions were exact in every repeat, 8 in none and 1 in some (a Context shape case, wrong in repeat 2). Jev can return different probabilities for the same request; the repeats show how much.",
        "Economics: recorded tokens of the benchmark runs repriced at list prices for each model mix (a calculation)."
      ],
      "caveats": [
        "The case sets and question wording were revised in fix waves against Jev answers on 2026-10-04 and 2026-10-05, so Jev has a home advantage.",
        "Jev’s numbers are from a live run: 3 repeats of the same 82 decisions, not 246 independent samples, so its interval is taken at n = 82. Its recorded production run of 2026-10-05 scored 74 of 82.",
        "Claude routers ran through the Claude Code CLI; the CLI adds startup time and tool-schema tokens a direct API call would not. Haiku 4.5 ran with the CLI default extended thinking; Sonnet 5.5 at effort low, as production asks.",
        "One sample per decision; production asks a second sample when confidence is low. Confidence-gated coverage is therefore not compared.",
        "82 cases in four small hand-labelled sets: intervals are wide.",
        "Clef / Clef-Flash (local): not measured (No local Clef server was running and installing a 6-20 GB model was out of scope for this run).",
        "Economics are calculations: Same tokens, same cache-read share and same number of calls on every model; a different model would take a different path and number of turns.",
        "Jev ran as a direct HTTPS call; the Claude routers ran through the Claude Code CLI. These are different routes, so speed and cost compare what a caller pays per decision, not one model against the other.",
        "Jev’s latency is one 35-second window from one Mac over a home network. The API reports no server time. A caller near the API would see less."
      ],
      "sourceIds": [
        "agent-routing",
        "calc-repricing",
        "price-jev",
        "price-anthropic",
        "agent-jev-live"
      ],
      "stats": [
        {
          "id": "exact-jev",
          "label": "Jev 1.13 (TypeSafe): exact decisions",
          "value": 0.8984,
          "unit": "rate",
          "display": "90% (74/82)",
          "n": 82,
          "ci": [
            0.8191,
            0.9497
          ],
          "note": "Live run: 3 repeats of the same 82 decisions. 221 of 246 calls were exact (89.8%; 74, 73 and 74 of 82 per repeat). The count is shown on the 82-decision scale (89.8% of 82 is 74), the scale of the interval: repeats of one decision are not independent, so the interval is taken at n = 82, not 246."
        },
        {
          "id": "exact-claude-haiku",
          "label": "Claude Haiku 4.5: exact decisions",
          "value": 0.8902,
          "unit": "rate",
          "display": "89% (73/82)",
          "n": 82,
          "ci": [
            0.8044,
            0.9412
          ]
        },
        {
          "id": "exact-claude-sonnet",
          "label": "Claude Sonnet 5.5: exact decisions",
          "value": 0.939,
          "unit": "rate",
          "display": "94% (77/82)",
          "n": 82,
          "ci": [
            0.8651,
            0.9737
          ]
        },
        {
          "id": "jev-cost-per-1000",
          "label": "Jev cost per 1,000 decisions",
          "value": 0.0337,
          "unit": "usd",
          "display": "$0.0337",
          "n": 246,
          "note": "Calculation: mean reported input tokens per decision × the published input price."
        },
        {
          "id": "economics-policy-vs-all-sonnet",
          "label": "Thought experiment: policy (Opus strong, Haiku ancillary) vs all Sonnet 5.5",
          "value": 1.489,
          "unit": "ratio",
          "display": "$161.62 (1.49x)",
          "n": 2362,
          "note": "Calculation on recorded tokens, not a run."
        },
        {
          "id": "economics-split-vs-all-sonnet",
          "label": "Thought experiment: split (Sonnet main line, Haiku ancillary) vs all Sonnet 5.5",
          "value": 0.9723,
          "unit": "ratio",
          "display": "$105.53 (0.97x)",
          "n": 2362,
          "note": "Calculation on recorded tokens, not a run."
        },
        {
          "id": "economics-cache-share",
          "label": "Cache-read share of recorded input (economics data)",
          "value": 0.9304,
          "unit": "rate",
          "display": "93.0%",
          "n": 2362
        }
      ],
      "charts": [
        {
          "id": "routing-exact-decisions",
          "title": "Typed routing decisions answered exactly right",
          "subtitle": "Share of asked cases where every scored question was acceptable",
          "kind": "dot-range",
          "unit": "rate",
          "yLabel": "Exact",
          "whisker": "ci95",
          "series": [
            {
              "name": "Exact rate",
              "points": [
                {
                  "label": "Jev 1.13 (TypeSafe)",
                  "value": 0.8984,
                  "lo": 0.8191,
                  "hi": 0.9497,
                  "n": 82,
                  "highlight": true
                },
                {
                  "label": "Claude Haiku 4.5",
                  "value": 0.8902,
                  "lo": 0.8044,
                  "hi": 0.9412,
                  "n": 82,
                  "highlight": false
                },
                {
                  "label": "Claude Sonnet 5.5",
                  "value": 0.939,
                  "lo": 0.8651,
                  "hi": 0.9737,
                  "n": 82,
                  "highlight": false
                }
              ]
            }
          ],
          "note": "Whiskers are 95% Wilson intervals on the 82 decisions. Jev is the live run: 3 repeats of the same 82 decisions. Repeats of one decision are not independent, so its interval is taken at n = 82, not 246. 221 of 246 Jev calls were exact (74, 73 and 74 of 82 per repeat). Each Claude router made one pass through the Claude Code CLI. The recorded production run of Jev scored 74 of 82. The case sets were revised against Jev answers, so Jev has a home advantage.",
          "sourceIds": [
            "agent-routing",
            "agent-jev-live"
          ]
        },
        {
          "id": "routing-key-accuracy",
          "title": "Per-question accuracy",
          "subtitle": "Each open question the router was asked; an unanswered question counts as wrong",
          "kind": "dot-range",
          "unit": "rate",
          "yLabel": "Correct answers",
          "whisker": "ci95",
          "series": [
            {
              "name": "Key accuracy",
              "points": [
                {
                  "label": "Jev 1.13 (TypeSafe)",
                  "value": 0.9485,
                  "lo": 0.9077,
                  "hi": 0.9718,
                  "n": 194,
                  "highlight": true
                },
                {
                  "label": "Claude Haiku 4.5",
                  "value": 0.9433,
                  "lo": 0.9013,
                  "hi": 0.968,
                  "n": 194,
                  "highlight": false
                },
                {
                  "label": "Claude Sonnet 5.5",
                  "value": 0.9742,
                  "lo": 0.9411,
                  "hi": 0.9889,
                  "n": 194,
                  "highlight": false
                }
              ]
            }
          ],
          "note": "Whiskers are 95% Wilson intervals on the 194 scored questions. Jev: 552 of 582 answers over 3 repeats, with the interval taken at n = 194 because the repeats are not independent. Each Claude router made one pass.",
          "sourceIds": [
            "agent-routing",
            "agent-jev-live"
          ]
        },
        {
          "id": "routing-exact-by-decision",
          "title": "Exact rate by decision type",
          "kind": "grouped-bar",
          "unit": "rate",
          "yLabel": "Exact",
          "series": [
            {
              "name": "Jev 1.13 (TypeSafe)",
              "points": [
                {
                  "label": "Failure class",
                  "value": 1,
                  "lo": 0.8241,
                  "hi": 1,
                  "n": 18,
                  "highlight": true
                },
                {
                  "label": "Message intent",
                  "value": 1,
                  "lo": 0.8389,
                  "hi": 1,
                  "n": 20,
                  "highlight": true
                },
                {
                  "label": "Is it a rule?",
                  "value": 1,
                  "lo": 0.7575,
                  "hi": 1,
                  "n": 12,
                  "highlight": true
                },
                {
                  "label": "Context shape",
                  "value": 0.7396,
                  "lo": 0.5789,
                  "hi": 0.8675,
                  "n": 32,
                  "highlight": true
                }
              ]
            },
            {
              "name": "Claude Haiku 4.5",
              "points": [
                {
                  "label": "Failure class",
                  "value": 0.9444,
                  "lo": 0.7424,
                  "hi": 0.9901,
                  "n": 18,
                  "highlight": false
                },
                {
                  "label": "Message intent",
                  "value": 1,
                  "lo": 0.8389,
                  "hi": 1,
                  "n": 20,
                  "highlight": false
                },
                {
                  "label": "Is it a rule?",
                  "value": 1,
                  "lo": 0.7575,
                  "hi": 1,
                  "n": 12,
                  "highlight": false
                },
                {
                  "label": "Context shape",
                  "value": 0.75,
                  "lo": 0.5789,
                  "hi": 0.8675,
                  "n": 32,
                  "highlight": false
                }
              ]
            },
            {
              "name": "Claude Sonnet 5.5",
              "points": [
                {
                  "label": "Failure class",
                  "value": 1,
                  "lo": 0.8241,
                  "hi": 1,
                  "n": 18,
                  "highlight": false
                },
                {
                  "label": "Message intent",
                  "value": 1,
                  "lo": 0.8389,
                  "hi": 1,
                  "n": 20,
                  "highlight": false
                },
                {
                  "label": "Is it a rule?",
                  "value": 1,
                  "lo": 0.7575,
                  "hi": 1,
                  "n": 12,
                  "highlight": false
                },
                {
                  "label": "Context shape",
                  "value": 0.8438,
                  "lo": 0.6825,
                  "hi": 0.9314,
                  "n": 32,
                  "highlight": false
                }
              ]
            }
          ],
          "note": "A missing bar means that router was not run on that decision type. Whiskers are 95% Wilson intervals. Jev: the pooled rate of the live repeats, with the interval taken at the number of decisions of that type.",
          "sourceIds": [
            "agent-routing",
            "agent-jev-live"
          ]
        },
        {
          "id": "routing-cost-per-1000",
          "title": "Cost per 1,000 routing decisions",
          "subtitle": "List price × reported tokens per decision",
          "kind": "bar",
          "unit": "usd",
          "yLabel": "USD per 1,000 decisions",
          "series": [
            {
              "name": "Cost",
              "points": [
                {
                  "label": "Jev 1.13 (TypeSafe)",
                  "value": 0.0337,
                  "n": 246,
                  "highlight": true
                },
                {
                  "label": "Claude Haiku 4.5",
                  "value": 8.924,
                  "n": 82,
                  "highlight": false
                },
                {
                  "label": "Claude Sonnet 5.5",
                  "value": 4.996,
                  "n": 82,
                  "highlight": false
                }
              ]
            }
          ],
          "note": "List-price calculation from tokens. Jev: the 803 input tokens per decision its API reported × the published $0.042 per million input tokens (output tokens are free); the provider-reported cost of its recorded run is the same figure. The LLM routers ran through a subscription CLI, so CLI tool-schema and thinking tokens are included because the CLI reports them.",
          "sourceIds": [
            "agent-routing",
            "calc-repricing",
            "price-jev",
            "price-anthropic",
            "agent-jev-live"
          ]
        },
        {
          "id": "routing-decision-latency",
          "title": "Time per routing decision",
          "subtitle": "Median wall time, whisker to the 95th percentile",
          "kind": "dot-range",
          "unit": "ms",
          "yLabel": "Time per decision",
          "series": [
            {
              "name": "Wall time (CLI)",
              "points": [
                {
                  "label": "Claude Haiku 4.5",
                  "value": 12674,
                  "lo": 12674,
                  "hi": 34413,
                  "n": 82
                },
                {
                  "label": "Claude Sonnet 5.5",
                  "value": 2598,
                  "lo": 2598,
                  "hi": 4298,
                  "n": 82
                }
              ]
            },
            {
              "name": "Wall time (direct API call)",
              "points": [
                {
                  "label": "Jev 1.13 (TypeSafe)",
                  "value": 136.5,
                  "lo": 136.5,
                  "hi": 195.7,
                  "n": 246
                }
              ]
            },
            {
              "name": "Model time (API)",
              "points": [
                {
                  "label": "Claude Haiku 4.5",
                  "value": 10734,
                  "lo": 10734,
                  "hi": 32072,
                  "n": 82
                },
                {
                  "label": "Claude Sonnet 5.5",
                  "value": 1599,
                  "lo": 1599,
                  "hi": 2574,
                  "n": 82
                }
              ]
            }
          ],
          "note": "Whiskers run from p50 to p95. The Claude routers ran through the Claude Code CLI, so their wall time includes CLI start-up and the tool schema; one pass of 82 decisions each. Jev was called directly over HTTPS from one Mac on a home network: 246 calls in a 35-second window, client wall time with the network inside it. Its API reports no server time, so Jev has no model-time point. These are different routes: the chart shows what a caller waits per decision, not model compute time.",
          "whisker": "p50-p95",
          "sourceIds": [
            "agent-routing",
            "agent-jev-live"
          ]
        },
        {
          "id": "routing-economics-scenarios",
          "title": "Thought experiment: recorded agent work under different model mixes",
          "subtitle": "50 benchmark runs, 2,362 model calls, repriced",
          "kind": "bar",
          "unit": "usd",
          "yLabel": "USD for all runs",
          "series": [
            {
              "name": "Repriced cost",
              "points": [
                {
                  "label": "all Fable 5.1",
                  "value": 369.58,
                  "highlight": false
                },
                {
                  "label": "all Opus 5.5",
                  "value": 170.92,
                  "highlight": false
                },
                {
                  "label": "policy (Opus strong, Haiku ancillary)",
                  "value": 161.62,
                  "highlight": false
                },
                {
                  "label": "all Sonnet 5.5",
                  "value": 108.54,
                  "highlight": true
                },
                {
                  "label": "split (Sonnet main line, Haiku ancillary)",
                  "value": 105.53,
                  "highlight": false
                },
                {
                  "label": "all Haiku 4.5",
                  "value": 54.27,
                  "highlight": false
                }
              ]
            }
          ],
          "note": "Calculation, not a run: every recorded call ran on Sonnet 5.5 with routing off. Same tokens on every model; a different model or mix would take a different path.",
          "sourceIds": [
            "agent-routing",
            "calc-repricing",
            "price-anthropic"
          ]
        },
        {
          "id": "routing-economics-by-stage",
          "title": "Thought experiment: repriced cost by pipeline stage",
          "kind": "grouped-bar",
          "unit": "usd",
          "yLabel": "USD",
          "series": [
            {
              "name": "Haiku 4.5",
              "points": [
                {
                  "label": "act (strong)",
                  "value": 28.66
                },
                {
                  "label": "research (strong)",
                  "value": 14.31
                },
                {
                  "label": "verify (strong)",
                  "value": 4.74
                },
                {
                  "label": "review (strong)",
                  "value": 3.36
                },
                {
                  "label": "memory and onboarding (economy)",
                  "value": 3.11
                },
                {
                  "label": "other (standard)",
                  "value": 0.1
                }
              ]
            },
            {
              "name": "Sonnet 5.5",
              "points": [
                {
                  "label": "act (strong)",
                  "value": 57.31
                },
                {
                  "label": "research (strong)",
                  "value": 28.62
                },
                {
                  "label": "verify (strong)",
                  "value": 9.48
                },
                {
                  "label": "review (strong)",
                  "value": 6.71
                },
                {
                  "label": "memory and onboarding (economy)",
                  "value": 6.22
                },
                {
                  "label": "other (standard)",
                  "value": 0.19
                }
              ]
            },
            {
              "name": "Opus 5.5",
              "points": [
                {
                  "label": "act (strong)",
                  "value": 78.86
                },
                {
                  "label": "research (strong)",
                  "value": 47.24
                },
                {
                  "label": "verify (strong)",
                  "value": 18.9
                },
                {
                  "label": "review (strong)",
                  "value": 13.21
                },
                {
                  "label": "memory and onboarding (economy)",
                  "value": 12.32
                },
                {
                  "label": "other (standard)",
                  "value": 0.39
                }
              ]
            },
            {
              "name": "Fable 5.1",
              "points": [
                {
                  "label": "act (strong)",
                  "value": 152.43
                },
                {
                  "label": "research (strong)",
                  "value": 105.58
                },
                {
                  "label": "verify (strong)",
                  "value": 47.18
                },
                {
                  "label": "review (strong)",
                  "value": 32.78
                },
                {
                  "label": "memory and onboarding (economy)",
                  "value": 30.65
                },
                {
                  "label": "other (standard)",
                  "value": 0.96
                }
              ]
            }
          ],
          "note": "Calculation, not a run. The tier in brackets is the routing policy tier for that stage.",
          "sourceIds": [
            "agent-routing",
            "calc-repricing",
            "price-anthropic"
          ]
        }
      ],
      "tables": [
        {
          "id": "routing-head-to-head",
          "title": "Same cases, two routers (Jev: its recorded production run, one pass)",
          "columns": [
            {
              "key": "pair",
              "label": "Pair",
              "unit": "text"
            },
            {
              "key": "bothRight",
              "label": "Both right",
              "unit": "count"
            },
            {
              "key": "onlyA",
              "label": "Only first right",
              "unit": "count"
            },
            {
              "key": "onlyB",
              "label": "Only second right",
              "unit": "count"
            },
            {
              "key": "bothWrong",
              "label": "Both wrong",
              "unit": "count"
            },
            {
              "key": "p",
              "label": "Exact McNemar p"
            }
          ],
          "rows": [
            {
              "pair": "Claude Haiku 4.5 vs Jev 1.13 (TypeSafe)",
              "bothRight": 70,
              "onlyA": 3,
              "onlyB": 4,
              "bothWrong": 5,
              "p": 1
            },
            {
              "pair": "Claude Sonnet 5.5 vs Jev 1.13 (TypeSafe)",
              "bothRight": 73,
              "onlyA": 4,
              "onlyB": 1,
              "bothWrong": 4,
              "p": 0.375
            }
          ]
        },
        {
          "id": "routing-question-accuracy",
          "title": "Per-question accuracy by decision type (Jev: its recorded production run, one pass)",
          "columns": [
            {
              "key": "decision",
              "label": "Decision",
              "unit": "text"
            },
            {
              "key": "question",
              "label": "Question",
              "unit": "text"
            },
            {
              "key": "jev",
              "label": "Jev 1.13 (TypeSafe)",
              "unit": "text"
            },
            {
              "key": "claude-haiku",
              "label": "Claude Haiku 4.5",
              "unit": "text"
            },
            {
              "key": "claude-sonnet",
              "label": "Claude Sonnet 5.5",
              "unit": "text"
            }
          ],
          "rows": [
            {
              "decision": "Failure class",
              "question": "failure",
              "jev": "18/18",
              "claude-haiku": "17/18",
              "claude-sonnet": "18/18"
            },
            {
              "decision": "Message intent",
              "question": "intent",
              "jev": "20/20",
              "claude-haiku": "20/20",
              "claude-sonnet": "20/20"
            },
            {
              "decision": "Is it a rule?",
              "question": "kind",
              "jev": "12/12",
              "claude-haiku": "12/12",
              "claude-sonnet": "12/12"
            },
            {
              "decision": "Context shape",
              "question": "turn",
              "jev": "5/6",
              "claude-haiku": "5/6",
              "claude-sonnet": "5/6"
            },
            {
              "decision": "Context shape",
              "question": "transcript",
              "jev": "12/16",
              "claude-haiku": "13/16",
              "claude-sonnet": "15/16"
            },
            {
              "decision": "Context shape",
              "question": "artifacts",
              "jev": "4/4",
              "claude-haiku": "4/4",
              "claude-sonnet": "4/4"
            },
            {
              "decision": "Context shape",
              "question": "knowledge",
              "jev": "22/24",
              "claude-haiku": "22/24",
              "claude-sonnet": "23/24"
            },
            {
              "decision": "Context shape",
              "question": "memories",
              "jev": "21/21",
              "claude-haiku": "21/21",
              "claude-sonnet": "20/21"
            },
            {
              "decision": "Context shape",
              "question": "examples",
              "jev": "10/10",
              "claude-haiku": "10/10",
              "claude-sonnet": "10/10"
            },
            {
              "decision": "Context shape",
              "question": "scope",
              "jev": "30/31",
              "claude-haiku": "29/31",
              "claude-sonnet": "31/31"
            },
            {
              "decision": "Context shape",
              "question": "complexity",
              "jev": "31/32",
              "claude-haiku": "30/32",
              "claude-sonnet": "31/32"
            }
          ]
        },
        {
          "id": "routing-routers",
          "title": "Every router",
          "columns": [
            {
              "key": "router",
              "label": "Router",
              "unit": "text"
            },
            {
              "key": "how",
              "label": "How measured",
              "unit": "text"
            },
            {
              "key": "exact",
              "label": "Exact",
              "unit": "text"
            },
            {
              "key": "keys",
              "label": "Key accuracy",
              "unit": "text"
            },
            {
              "key": "tokens",
              "label": "Tokens per decision (input incl. cache / output)",
              "unit": "text"
            },
            {
              "key": "cost",
              "label": "USD per 1,000",
              "unit": "usd"
            }
          ],
          "rows": [
            {
              "router": "Jev 1.13 (TypeSafe)",
              "how": "live API run 2026-10-06: 3 repeats of the same 82 decisions, 246 counted calls one at a time from one Mac. All calls: 221 of 246 exact and 552 of 582 questions; the counts shown are on the 82-decision scale of the interval. The recorded production run of 2026-10-05 scored 74 of 82. Cost is a calculation from the reported input tokens",
              "exact": "74/82",
              "keys": "184/194",
              "tokens": "803 / 148",
              "cost": 0.0337
            },
            {
              "router": "Claude Haiku 4.5",
              "how": "this benchmark, Claude Code CLI on a subscription account, one call per decision",
              "exact": "73/82",
              "keys": "183/194",
              "tokens": "1,831 / 1,419",
              "cost": 8.924
            },
            {
              "router": "Claude Sonnet 5.5",
              "how": "this benchmark, Claude Code CLI on a subscription account, one call per decision",
              "exact": "77/82",
              "keys": "189/194",
              "tokens": "1,785 / 107",
              "cost": 4.996
            },
            {
              "router": "Clef / Clef-Flash (local)",
              "how": "not measured: No local Clef server was running and installing a 6-20 GB model was out of scope for this run.",
              "exact": "not measured",
              "keys": "not measured",
              "tokens": "",
              "cost": null
            }
          ]
        },
        {
          "id": "routing-jev-live-repeats",
          "title": "Jev live run, repeat by repeat",
          "columns": [
            {
              "key": "run",
              "label": "Run",
              "unit": "text"
            },
            {
              "key": "exact",
              "label": "Exact decisions",
              "unit": "text"
            },
            {
              "key": "keys",
              "label": "Questions answered acceptably",
              "unit": "text"
            },
            {
              "key": "medianMs",
              "label": "Median time per call (ms)",
              "unit": "ms"
            }
          ],
          "rows": [
            {
              "run": "Repeat 1 (without the cold first call)",
              "exact": "74/82",
              "keys": "184/194",
              "medianMs": 130.3
            },
            {
              "run": "Repeat 2",
              "exact": "73/82",
              "keys": "183/194",
              "medianMs": 141.7
            },
            {
              "run": "Repeat 3",
              "exact": "74/82",
              "keys": "185/194",
              "medianMs": 137.1
            },
            {
              "run": "All 3 repeats (246 calls)",
              "exact": "221/246 = 89.8%; case-level 95% interval 81.9% to 95.0%",
              "keys": "552/582 = 94.8%; 90.8% to 97.2%",
              "medianMs": 136.5
            },
            {
              "run": "Recorded production run, 2026-10-05 (one pass, no latency recorded)",
              "exact": "74/82",
              "keys": "185/194",
              "medianMs": null
            }
          ]
        }
      ]
    },
    {
      "slug": "routing-overhead",
      "title": "Routing overhead: deterministic policy vs LLM routers vs Jev",
      "seoTitle": "Routing overhead: rules vs LLM routers vs Jev",
      "description": "How much delay and cost a router adds per decision: an in-process policy, Claude routers through a CLI, and Jev. Plus CLI start-up tax and per-task totals.",
      "question": "What delay and what cost does each kind of router add before the real work of a call starts?",
      "answer": "The deterministic routing policy decided in a median 1.42 µs (p95 2.33 µs, 20,000 decisions, $0). The fastest LLM router, Claude Sonnet 5.5 through the Claude Code CLI, took a median 2.60 s per decision (p95 4.30 s, n = 82), about 1.8 million times longer; 973 ms of that median was CLI time, not model time. Haiku 4.5 with its default thinking took 12.54 s (p95 34.48 s). Jev 1.13, called directly over HTTPS from the same Mac, took a median 137 ms per decision (p95 196 ms, n = 246 calls, client wall time with the network inside it) and cost $0.0337 per 1,000 decisions (a calculation from its reported input tokens). That is a different route from the CLI routers, so it is not a model-against-model comparison. As a calculation over 48 recorded tasks (median 49.5 model calls each), routing every call would add $1.67 per 1,000 tasks and up to 7 s of waiting per task with Jev, $247.30 with Sonnet (8.2% of the work cost) and up to 129 s of waiting per task with Sonnet; routing only the 7 System One decisions cuts Sonnet to $34.97 and 18 s. CLI start-up alone, for a one-word answer: Claude Code (Haiku 4.5) took 2.53 s and sent 6,761 input tokens; Codex CLI took 6.00 s and sent 17,051 input tokens, 13,184 of them read from the cache (5 runs each, different models).",
      "date": "2026-10-06",
      "updated": "2026-10-06",
      "tags": [
        "routing",
        "latency",
        "overhead",
        "jev",
        "llm-router",
        "cli",
        "calculation"
      ],
      "method": [
        "Protocol declared before any measurement.",
        "Deterministic policy: the platform’s production routing decision (plan, model ladder and effort) on its default policy, timed in process on one Apple M3 Ultra Mac with Node 25: 5,000 warm-up calls, then 20,000 timed decisions over 64 synthetic routing contexts, plus a 200,000-call batch for throughput. Database reads and the decision record write are out of scope; the recorded rule-based System One decisions (millisecond resolution, record write included) are shown as a stat.",
        "LLM routers: the recorded routing runs of the routing study (Claude Sonnet 5.5 (effort low, via Claude Code), 82 calls; Claude Haiku 4.5 (thinking on, via Claude Code), 82 calls), one call at a time. Wall time per call, the API time the CLI reported, and the difference (CLI and harness time). p50 and p95 recomputed from the per-call log.",
        "Jev: a live run on 2026-10-06. The same 82 typed decisions as the routing study, 3 repeats, 246 counted calls sent one at a time over HTTPS to the TypeSafe API from the same Mac, on a home network. Time per call is client wall time from before the request to after the body was read, so the network is inside it; the API reports no server time. Cost per decision is a calculation: the mean input tokens its API reported × the published price. The recorded production run (82 decisions) has the same cost as a provider-reported figure.",
        "CLI start-up: Claude Code · Claude Haiku 4.5 5 runs, Codex CLI (default model) 5 runs, a one-word prompt, one call at a time. Times to the first output event, the first model output and exit.",
        "Per 1,000 tasks (a calculation): decisions per task from 48 recorded bench runs read only (2 empty runs excluded) × cost and median time per decision. Decisions are assumed to wait in line, so the delay is an upper bound."
      ],
      "caveats": [
        "The policy is timed in process and the LLM routers through a CLI: this compares the two ways of routing as deployed, not two models on equal footing. A direct API call would skip the CLI time (shown separately).",
        "The routing study reports its latency medians from its own summary; this study recomputes them from the per-call log, so the medians can differ by a few tens of milliseconds.",
        "Haiku 4.5 ran with the CLI’s default thinking, which makes it slower than Sonnet 5.5 at effort low here.",
        "OpenRouter’s Auto Router and cheaper hosted inference were not timed: no key in the environment. A local router server was not running, so it was not timed either.",
        "CLI timings come from one Mac with 5 runs per CLI; a range is not a confidence interval. Codex CLI reports no API time, so its CLI time cannot be separated from model time.",
        "Per-task numbers are calculations on runs where routing was off; the work cost is the recorded list-price estimate for Sonnet 5.5.",
        "Jev was timed over a direct HTTPS call; the Claude routers ran through the CLI. These are different routes, so the gap is what a caller waits per decision, not model compute time. A caller closer to the API would see less than this Mac on a home network did.",
        "Jev’s timing is one 35-second window on 2026-10-06, 246 calls from one machine; server load at that time is unknown. The 246 calls are 3 repeats of 82 requests, so they are not independent draws.",
        "Corrected 2026-10-07: an earlier version of this page counted the cached tokens twice (30,235). The CLI reports 17,051 input tokens including 13,184 read from the cache."
      ],
      "sourceIds": [
        "agent-routing-overhead",
        "calc-routing-overhead",
        "agent-routing",
        "price-anthropic",
        "price-jev",
        "agent-jev-live"
      ],
      "stats": [
        {
          "id": "router-overhead-policy-p50",
          "label": "Deterministic routing policy: median decision time",
          "value": 0.00142,
          "unit": "ms",
          "display": "1.42 µs (p95 2.33 µs, p99 3.04 µs)",
          "n": 20000,
          "note": "Timer resolution 0.041 µs; 469,409 decisions per second in a 200,000-call batch."
        },
        {
          "id": "router-overhead-speedup",
          "label": "Median LLM router call ÷ median policy decision",
          "value": 1828873,
          "unit": "ratio",
          "display": "about 1.8 million times",
          "n": 82
        },
        {
          "id": "router-overhead-system-one-rule-arm",
          "label": "Recorded rule-based System One decision, record write included",
          "value": 1,
          "unit": "ms",
          "display": "1 ms median, 2 ms p95 (millisecond resolution)",
          "n": 419
        },
        {
          "id": "router-overhead-decisions-per-task",
          "label": "Model calls per task (each one a routing decision)",
          "value": 49.5,
          "unit": "calls",
          "display": "49.5 median (13 to 73)",
          "n": 48
        },
        {
          "id": "router-overhead-jev-p50",
          "label": "Jev 1.13 (TypeSafe): median decision time over the API",
          "value": 136.5,
          "unit": "ms",
          "display": "137 ms (p95 196 ms)",
          "n": 246,
          "note": "Client wall time from one Mac over a home network, 246 calls in a 35-second window. The API reports no server time."
        },
        {
          "id": "router-overhead-sonnet-vs-jev",
          "label": "Median Claude Sonnet 5.5 call through the CLI ÷ median Jev call over the API",
          "value": 19,
          "unit": "ratio",
          "display": "about 19 times",
          "n": 82,
          "note": "A calculation across two routes (CLI vs a direct API call from one Mac), not a model-against-model comparison."
        },
        {
          "id": "router-overhead-haiku-vs-jev",
          "label": "Median Claude Haiku 4.5 call through the CLI ÷ median Jev call over the API",
          "value": 91.9,
          "unit": "ratio",
          "display": "about 92 times",
          "n": 82,
          "note": "A calculation across two routes (CLI vs a direct API call from one Mac), not a model-against-model comparison."
        },
        {
          "id": "cli-startup-claude-harness-ms",
          "label": "Claude Code time outside the model on a one-word answer",
          "value": 1690,
          "unit": "ms",
          "display": "1,690 ms median (1,533 to 1,811)",
          "n": 5
        }
      ],
      "charts": [
        {
          "id": "router-overhead-decision-latency",
          "title": "Time to make one routing decision",
          "subtitle": "Median; whiskers = median to 95th percentile",
          "kind": "dot-range",
          "unit": "ms",
          "yLabel": "Time per decision",
          "whisker": "p50-p95",
          "series": [
            {
              "name": "Decision time",
              "points": [
                {
                  "label": "Deterministic routing policy (Agent, in process)",
                  "value": 0.00142,
                  "lo": 0.00142,
                  "hi": 0.00233,
                  "n": 20000,
                  "highlight": true
                },
                {
                  "label": "Jev 1.13 (TypeSafe)",
                  "value": 136.5,
                  "lo": 136.5,
                  "hi": 195.7,
                  "n": 246
                },
                {
                  "label": "Claude Sonnet 5.5 (effort low, via Claude Code)",
                  "value": 2597,
                  "lo": 2597,
                  "hi": 4298,
                  "n": 82
                },
                {
                  "label": "Claude Haiku 4.5 (thinking on, via Claude Code)",
                  "value": 12543,
                  "lo": 12543,
                  "hi": 34481,
                  "n": 82
                }
              ]
            }
          ],
          "note": "The policy is the pure in-process decision (median 1.42 µs, p95 2.33 µs), timed over 20,000 decisions after 5,000 warm-up calls. The Claude routers are wall time per call through the Claude Code CLI, one call at a time, from the recorded routing runs. Jev is wall time of a direct HTTPS call from the same Mac over a home network (246 calls in a 35-second window; its API reports no server time), so the network is inside it. A different route from the CLI, so the chart shows what a caller waits, not model compute time. The whisker is the median to the 95th percentile, not a confidence interval.",
          "sourceIds": [
            "agent-routing-overhead",
            "agent-routing",
            "agent-jev-live"
          ]
        },
        {
          "id": "router-overhead-cli-vs-model-time",
          "title": "Where an LLM router’s time goes: model vs CLI",
          "subtitle": "Median per call; whiskers = median to 95th percentile",
          "kind": "dot-range",
          "unit": "ms",
          "yLabel": "Time per call",
          "whisker": "p50-p95",
          "series": [
            {
              "name": "Model API time",
              "points": [
                {
                  "label": "Claude Sonnet 5.5 (effort low, via Claude Code)",
                  "value": 1596,
                  "lo": 1596,
                  "hi": 2583,
                  "n": 82
                },
                {
                  "label": "Claude Haiku 4.5 (thinking on, via Claude Code)",
                  "value": 10508,
                  "lo": 10508,
                  "hi": 32132,
                  "n": 82
                }
              ]
            },
            {
              "name": "CLI and harness time",
              "points": [
                {
                  "label": "Claude Sonnet 5.5 (effort low, via Claude Code)",
                  "value": 973,
                  "lo": 973,
                  "hi": 1277,
                  "n": 82
                },
                {
                  "label": "Claude Haiku 4.5 (thinking on, via Claude Code)",
                  "value": 1698,
                  "lo": 1698,
                  "hi": 2677,
                  "n": 82
                }
              ]
            }
          ],
          "note": "Model API time is the API duration the CLI reports; CLI and harness time is wall time minus that, per call. Medians of the parts do not add up to the median of the whole. Jev is not split: its API reports no server time. The whisker is the median to the 95th percentile, not a confidence interval.",
          "sourceIds": [
            "agent-routing-overhead",
            "agent-routing"
          ]
        },
        {
          "id": "router-overhead-completed",
          "title": "Routing calls that returned a decision",
          "subtitle": "Completed calls ÷ calls; whiskers = 95% Wilson interval",
          "kind": "dot-range",
          "unit": "rate",
          "yLabel": "Completed",
          "whisker": "ci95",
          "series": [
            {
              "name": "Completed",
              "points": [
                {
                  "label": "Deterministic routing policy (Agent, in process)",
                  "value": 1,
                  "lo": 0.9998,
                  "hi": 1,
                  "n": 20000,
                  "highlight": true
                },
                {
                  "label": "Jev 1.13 (TypeSafe)",
                  "value": 1,
                  "lo": 0.9846,
                  "hi": 1,
                  "n": 246
                },
                {
                  "label": "Claude Sonnet 5.5 (effort low, via Claude Code)",
                  "value": 1,
                  "lo": 0.9552,
                  "hi": 1,
                  "n": 82
                },
                {
                  "label": "Claude Haiku 4.5 (thinking on, via Claude Code)",
                  "value": 1,
                  "lo": 0.9552,
                  "hi": 1,
                  "n": 82
                }
              ]
            }
          ],
          "note": "A completed call returned a decision, right or wrong (accuracy is in the routing study). Whiskers are 95% Wilson intervals.",
          "sourceIds": [
            "agent-routing-overhead",
            "agent-routing",
            "agent-jev-live"
          ]
        },
        {
          "id": "router-overhead-cost-reported",
          "title": "Cost per 1,000 routing decisions: no model call vs provider-reported",
          "subtitle": "USD per 1,000 decisions",
          "kind": "bar",
          "unit": "usd",
          "yLabel": "USD per 1,000 decisions",
          "series": [
            {
              "name": "Cost per 1,000 decisions",
              "points": [
                {
                  "label": "Deterministic routing policy (Agent, in process)",
                  "value": 0,
                  "n": 20000,
                  "highlight": true
                },
                {
                  "label": "Jev 1.13 (TypeSafe)",
                  "value": 0.0337,
                  "n": 82,
                  "highlight": false
                }
              ]
            }
          ],
          "note": "The policy makes no model call, so it costs nothing per decision. Jev’s figure is the cost its provider reported for the recorded production run (82 decisions). The input tokens of the live run give the same figure at the published price. The Claude routers are in the next chart: their cost is derived from list prices.",
          "sourceIds": [
            "agent-routing-overhead",
            "agent-routing",
            "price-jev"
          ]
        },
        {
          "id": "router-overhead-cost-list-price",
          "title": "Cost per 1,000 routing decisions for the model routers (calculation)",
          "subtitle": "List price × the tokens each route reported, USD per 1,000 decisions",
          "kind": "bar",
          "unit": "usd",
          "yLabel": "USD per 1,000 decisions",
          "series": [
            {
              "name": "Cost per 1,000 decisions (list price)",
              "points": [
                {
                  "label": "Jev 1.13 (TypeSafe)",
                  "value": 0.0337,
                  "n": 246
                },
                {
                  "label": "Claude Sonnet 5.5 (effort low, via Claude Code)",
                  "value": 4.996,
                  "n": 82
                },
                {
                  "label": "Claude Haiku 4.5 (thinking on, via Claude Code)",
                  "value": 8.924,
                  "n": 82
                }
              ]
            }
          ],
          "note": "A calculation, not a bill: the Claude calls ran on a subscription. Jev: the 803 input tokens per decision its API reported × the published $0.042 per million input tokens (output tokens are free). Same tokens and prices as the routing study.",
          "sourceIds": [
            "agent-routing-overhead",
            "calc-routing-overhead",
            "agent-routing",
            "price-anthropic",
            "price-jev",
            "agent-jev-live"
          ]
        },
        {
          "id": "router-overhead-cost-per-1000-tasks",
          "title": "Added routing cost per 1,000 tasks (calculation)",
          "subtitle": "Decisions per task from recorded runs × cost per decision",
          "kind": "grouped-bar",
          "unit": "usd",
          "yLabel": "USD per 1,000 tasks",
          "series": [
            {
              "name": "Every model call routed (49.5 per task)",
              "points": [
                {
                  "label": "Deterministic routing policy (Agent, in process)",
                  "value": 0
                },
                {
                  "label": "Jev 1.13 (TypeSafe)",
                  "value": 1.67
                },
                {
                  "label": "Claude Sonnet 5.5 (effort low, via Claude Code)",
                  "value": 247.3
                },
                {
                  "label": "Claude Haiku 4.5 (thinking on, via Claude Code)",
                  "value": 441.74
                }
              ]
            },
            {
              "name": "Only System One decisions (7 per task)",
              "points": [
                {
                  "label": "Deterministic routing policy (Agent, in process)",
                  "value": 0
                },
                {
                  "label": "Jev 1.13 (TypeSafe)",
                  "value": 0.24
                },
                {
                  "label": "Claude Sonnet 5.5 (effort low, via Claude Code)",
                  "value": 34.97
                },
                {
                  "label": "Claude Haiku 4.5 (thinking on, via Claude Code)",
                  "value": 62.47
                }
              ]
            }
          ],
          "note": "A calculation. Decisions per task: the median of 48 recorded bench runs (routing was off in them, so every model call counts as one decision a router would make). Median recorded work cost per task: $3.03. Claude router costs are list-price calculations; Jev’s is a list-price calculation too (its recorded run’s provider-reported cost is the same).",
          "sourceIds": [
            "agent-routing-overhead",
            "calc-routing-overhead",
            "agent-routing",
            "price-anthropic",
            "price-jev",
            "agent-jev-live"
          ]
        },
        {
          "id": "router-overhead-delay-per-task",
          "title": "Added routing delay per task (calculation)",
          "subtitle": "Decisions per task × median decision time, if every decision waits in line",
          "kind": "grouped-bar",
          "unit": "seconds",
          "yLabel": "Seconds per task",
          "series": [
            {
              "name": "Every model call routed (49.5 per task)",
              "points": [
                {
                  "label": "Deterministic routing policy (Agent, in process)",
                  "value": 0.0000703
                },
                {
                  "label": "Jev 1.13 (TypeSafe)",
                  "value": 6.7568
                },
                {
                  "label": "Claude Sonnet 5.5 (effort low, via Claude Code)",
                  "value": 128.5515
                },
                {
                  "label": "Claude Haiku 4.5 (thinking on, via Claude Code)",
                  "value": 620.8785
                }
              ]
            },
            {
              "name": "Only System One decisions (7 per task)",
              "points": [
                {
                  "label": "Deterministic routing policy (Agent, in process)",
                  "value": 0.0000099
                },
                {
                  "label": "Jev 1.13 (TypeSafe)",
                  "value": 0.9555
                },
                {
                  "label": "Claude Sonnet 5.5 (effort low, via Claude Code)",
                  "value": 18.179
                },
                {
                  "label": "Claude Haiku 4.5 (thinking on, via Claude Code)",
                  "value": 87.801
                }
              ]
            }
          ],
          "note": "A calculation and an upper bound: it assumes each decision waits for the one before. Median recorded task wall time: 10.3 min. Jev’s delay uses its live median over the API from one Mac (network included); the Claude routers’ includes the CLI.",
          "sourceIds": [
            "agent-routing-overhead",
            "calc-routing-overhead",
            "agent-routing",
            "price-anthropic",
            "price-jev",
            "agent-jev-live"
          ]
        },
        {
          "id": "cli-startup-tax",
          "title": "CLI start-up tax on a one-word answer",
          "subtitle": "Median of 5 runs; whiskers = fastest and slowest run",
          "kind": "dot-range",
          "unit": "ms",
          "yLabel": "Time to output",
          "whisker": "minmax",
          "series": [
            {
              "name": "First output event",
              "points": [
                {
                  "label": "Claude Code · Claude Haiku 4.5",
                  "value": 563,
                  "lo": 519,
                  "hi": 726,
                  "n": 5
                },
                {
                  "label": "Codex CLI (default model)",
                  "value": 489,
                  "lo": 354,
                  "hi": 1304,
                  "n": 5
                }
              ]
            },
            {
              "name": "First model output",
              "points": [
                {
                  "label": "Claude Code · Claude Haiku 4.5",
                  "value": 1461,
                  "lo": 1206,
                  "hi": 2308,
                  "n": 5
                },
                {
                  "label": "Codex CLI (default model)",
                  "value": 5059,
                  "lo": 4391,
                  "hi": 5478,
                  "n": 5
                }
              ]
            },
            {
              "name": "Total wall time",
              "points": [
                {
                  "label": "Claude Code · Claude Haiku 4.5",
                  "value": 2529,
                  "lo": 2273,
                  "hi": 3382,
                  "n": 5
                },
                {
                  "label": "Codex CLI (default model)",
                  "value": 5999,
                  "lo": 5367,
                  "hi": 6506,
                  "n": 5
                }
              ]
            }
          ],
          "note": "Prompt: reply with one word. Claude Code · Claude Haiku 4.5: 5/5 runs completed; Codex CLI (default model): 5/5 runs completed. Isolated flags (no tools, no MCP servers, no session) for Claude Code; read-only sandbox and a fresh folder for Codex. The two CLIs ran different models, so CLI and model are not separated. A range, not a confidence interval.",
          "sourceIds": [
            "agent-routing-overhead"
          ]
        },
        {
          "id": "cli-startup-input-tokens",
          "title": "Input tokens a CLI sends for a one-word answer",
          "subtitle": "Per call, mostly the CLI’s own system prompt and tool definitions",
          "kind": "bar",
          "unit": "tokens",
          "yLabel": "Input tokens per call",
          "series": [
            {
              "name": "Input tokens per call",
              "points": [
                {
                  "label": "Claude Code · Claude Haiku 4.5",
                  "value": 6761,
                  "n": 5
                },
                {
                  "label": "Codex CLI (default model)",
                  "value": 17051,
                  "n": 5
                }
              ]
            }
          ],
          "note": "Claude Code sums its disjoint input, cache-read and cache-write fields. Codex CLI reports 17,051 input tokens, 13,184 of them read from the cache. The prompt itself is a few tokens.",
          "sourceIds": [
            "agent-routing-overhead"
          ]
        }
      ],
      "tables": [
        {
          "id": "router-overhead-status",
          "title": "What was measured, recorded, calculated or not measured",
          "columns": [
            {
              "key": "router",
              "label": "Router",
              "unit": "text"
            },
            {
              "key": "kind",
              "label": "Kind",
              "unit": "text"
            },
            {
              "key": "latency",
              "label": "Latency",
              "unit": "text"
            },
            {
              "key": "cost",
              "label": "Cost",
              "unit": "text"
            }
          ],
          "rows": [
            {
              "router": "Deterministic routing policy (Agent, in process)",
              "kind": "in-process rules",
              "latency": "measured 2026-10-06: 1.42 µs median",
              "cost": "$0 (no model call)"
            },
            {
              "router": "Jev 1.13 (TypeSafe)",
              "kind": "hosted decision model",
              "latency": "measured 2026-10-06: 137 ms median, p95 196 ms (246 calls over the API from one Mac; client wall time, no server time)",
              "cost": "$0.0337 per 1,000 (list-price calculation from input tokens; the recorded run’s own cost figure agrees)"
            },
            {
              "router": "Claude Sonnet 5.5 (effort low, via Claude Code)",
              "kind": "LLM router via agent CLI",
              "latency": "recorded: 2.60 s median (82 calls)",
              "cost": "$4.996 per 1,000 (list-price calculation)"
            },
            {
              "router": "Claude Haiku 4.5 (thinking on, via Claude Code)",
              "kind": "LLM router via agent CLI",
              "latency": "recorded: 12.54 s median (82 calls)",
              "cost": "$8.924 per 1,000 (list-price calculation)"
            },
            {
              "router": "OpenRouter Auto Router / cheaper hosted inference",
              "kind": "hosted LLM router / gateway",
              "latency": "Not measured: no key in the environment. The harness is ready and runs when a key is set.",
              "cost": "Not measured: no key in the environment. The harness is ready and runs when a key is set."
            },
            {
              "router": "Clef / Clef-Flash (local)",
              "kind": "local router model",
              "latency": "Not measured: no local server running.",
              "cost": "Not measured: no local server running."
            }
          ]
        },
        {
          "id": "router-overhead-jev-live",
          "title": "Jev live run: time per call and cost",
          "columns": [
            {
              "key": "measure",
              "label": "Measure",
              "unit": "text"
            },
            {
              "key": "value",
              "label": "Value",
              "unit": "text"
            }
          ],
          "rows": [
            {
              "measure": "Counted calls",
              "value": "246 (3 repeats of the same 82 decisions, one at a time), 0 failed"
            },
            {
              "measure": "Median time per call",
              "value": "136.5 ms"
            },
            {
              "measure": "90th percentile",
              "value": "171.0 ms"
            },
            {
              "measure": "95th percentile",
              "value": "195.7 ms"
            },
            {
              "measure": "Fastest and slowest call",
              "value": "100.9 ms and 297.3 ms"
            },
            {
              "measure": "Cold first call (new process, fresh connection)",
              "value": "224.7 ms"
            },
            {
              "measure": "Median per repeat",
              "value": "130.3 ms, 141.7 ms, 137.1 ms (the first repeat without the cold call)"
            },
            {
              "measure": "Input and output tokens per decision (mean)",
              "value": "803 and 148 (output tokens are free at the published price)"
            },
            {
              "measure": "Cost per 1,000 decisions (calculation)",
              "value": "$0.0337 = 803 input tokens × 1,000 × $0.042 per million"
            },
            {
              "measure": "Server-side time",
              "value": "not available: the API sends no timing header or field"
            }
          ]
        }
      ],
      "related": [
        "routing-jev-vs-llm",
        "cli-model-latency-tokens",
        "inference-provider-index"
      ]
    },
    {
      "slug": "inference-provider-index",
      "title": "Inference provider index: 27 models, 52 providers",
      "seoTitle": "Inference provider prices: OpenRouter vs direct",
      "description": "Price per million tokens for 27 models across 52 providers, the spread between them and OpenRouter’s markup over first-party prices.",
      "question": "For the same model, how much do inference providers differ in price, and what does a gateway such as OpenRouter add over the first-party list price?",
      "answer": "OpenRouter’s public API listed 265 endpoints from 52 providers for 27 models on 2026-10-06. For 10 of the 10 models with a first-party list price, OpenRouter’s per-token price was the same as the vendor’s; the cost of the gateway is the 5.5% credit-purchase fee (Standard plan, third-party-reported), so the effective markup is +5.5%. 15 of 15 closed models (Claude, GPT, Gemini) with 2 or more providers had one price across every standard-tier provider; their price differences come from named tiers (flex, priority, fast) and regional endpoints. Open-weight models differ by provider: DeepSeek V4 Flash 0423 12.6x (15 providers), DeepSeek V4 Pro 0423 11.2x (15 providers), gpt-oss-120b 6.9x (20 providers), Llama 3.3 70B Instruct 6.7x (10 providers) between the most expensive and the cheapest standard-tier provider (a calculation on reported prices). The cheapest endpoints often report lower precision or a shorter context. Latency and throughput were not in the keyless API (0 of 265 endpoints), and the gateway’s own delay is not measured: no key in the environment.",
      "date": "2026-10-06",
      "updated": "2026-10-06",
      "tags": [
        "inference",
        "providers",
        "openrouter",
        "pricing",
        "gateway",
        "open-weight",
        "third-party-reported"
      ],
      "method": [
        "One snapshot of OpenRouter’s public, keyless API on 2026-10-06: the model list and the endpoint list of 27 curated models (28 requests). Prices, context, quantization and uptime are copied as the API reported them; a field it did not return stays empty.",
        "Tier: read from the endpoint tag suffix. Flex, priority, fast, ultrafast and batch are named tiers; a region suffix (us, eu, europe, a cloud region) is regional; anything else is standard. This is a heuristic.",
        "Per-model charts show the standard tier only, with one bar per provider: its cheapest standard endpoint. The endpoint table lists every endpoint in every tier.",
        "Blended price = (3 × input + output) ÷ 4, a 3:1 input:output token mix. Spread = most expensive ÷ cheapest blended price among standard-tier providers. Both are calculations on reported prices.",
        "Markup = OpenRouter list price ÷ first-party list price − 1. First-party prices are the vendors’ published list prices as recorded in the product price table. The effective markup adds the credit-purchase fee for card purchases on the Standard plan.",
        "Gateway delay (time to first token, total time, billed cost per call): not measured: no key in the environment. A live harness is ready; it sends nothing without a key and has a hard spending cap."
      ],
      "caveats": [
        "Every price is third-party-reported by OpenRouter’s API at 2026-10-06. Prices change often; refetch before relying on them.",
        "A provider listed on OpenRouter is reached through OpenRouter; its price there may differ from the price on the provider’s own site.",
        "The cheapest endpoint may run lower precision (fp4 or fp8) or a shorter context. Price alone does not make two endpoints equal.",
        "No latency or throughput figure: the keyless API returned none. A cheaper provider is not shown to be slower or faster.",
        "First-party prices in the product table were verified on an earlier date than the snapshot; a vendor price change in between would show as a markup.",
        "Comparison rows between providers name no winner: a price has no interval, so the rule for winners does not apply. The gap is stated."
      ],
      "sourceIds": [
        "openrouter-api-snapshot",
        "openrouter-fees",
        "price-anthropic",
        "price-openai",
        "price-google"
      ],
      "stats": [
        {
          "id": "provider-index-endpoints",
          "label": "Provider endpoints in the snapshot",
          "value": 265,
          "unit": "count",
          "display": "265 endpoints, 52 providers, 27 models",
          "n": 27
        },
        {
          "id": "provider-index-max-spread",
          "label": "Largest standard-tier price spread (DeepSeek V4 Flash 0423)",
          "value": 12.57,
          "unit": "ratio",
          "display": "12.6x (most expensive: Cloudflare; cheapest: StreamLake (fp8))",
          "n": 15,
          "note": "Calculation on reported prices, blended 3:1."
        },
        {
          "id": "gateway-zero-markup-models",
          "label": "Models where OpenRouter’s per-token price equals the first-party list price",
          "value": 10,
          "unit": "count",
          "display": "10 of 10",
          "n": 10
        },
        {
          "id": "gateway-credit-fee",
          "label": "Credit-purchase fee on OpenRouter’s Standard plan (third-party-reported)",
          "value": 5.5,
          "unit": "percent",
          "display": "5.5% ($0.80 minimum by card)"
        },
        {
          "id": "provider-index-latency-reported",
          "label": "Endpoints with a latency figure in the keyless API",
          "value": 0,
          "unit": "count",
          "display": "0 of 265",
          "n": 265
        }
      ],
      "charts": [
        {
          "id": "provider-index-spread",
          "title": "How much more the priciest provider charges than the cheapest",
          "subtitle": "Standard tier, blended price (3 input : 1 output), most expensive provider ÷ cheapest provider",
          "kind": "bar",
          "unit": "ratio",
          "yLabel": "Most expensive ÷ cheapest",
          "series": [
            {
              "name": "Price spread",
              "points": [
                {
                  "label": "DeepSeek V4 Flash 0423",
                  "value": 12.57,
                  "n": 15,
                  "highlight": true
                },
                {
                  "label": "DeepSeek V4 Pro 0423",
                  "value": 11.21,
                  "n": 15,
                  "highlight": true
                },
                {
                  "label": "gpt-oss-120b",
                  "value": 6.92,
                  "n": 20,
                  "highlight": true
                },
                {
                  "label": "Llama 3.3 70B Instruct",
                  "value": 6.71,
                  "n": 10,
                  "highlight": true
                },
                {
                  "label": "GLM 5.3",
                  "value": 4.69,
                  "n": 32,
                  "highlight": true
                },
                {
                  "label": "Kimi K3",
                  "value": 1.78,
                  "n": 19,
                  "highlight": false
                },
                {
                  "label": "Llama 4 Maverick",
                  "value": 1.69,
                  "n": 3,
                  "highlight": false
                },
                {
                  "label": "Claude Haiku 4.5",
                  "value": 1,
                  "n": 4,
                  "highlight": false
                },
                {
                  "label": "Claude Sonnet 5",
                  "value": 1,
                  "n": 5,
                  "highlight": false
                },
                {
                  "label": "Claude Sonnet 5.5",
                  "value": 1,
                  "n": 5,
                  "highlight": false
                },
                {
                  "label": "Claude Opus 4.8",
                  "value": 1,
                  "n": 5,
                  "highlight": false
                },
                {
                  "label": "Claude Opus 5",
                  "value": 1,
                  "n": 5,
                  "highlight": false
                },
                {
                  "label": "Claude Opus 5.5",
                  "value": 1,
                  "n": 5,
                  "highlight": false
                },
                {
                  "label": "Claude Fable 5.1",
                  "value": 1,
                  "n": 4,
                  "highlight": false
                },
                {
                  "label": "GPT-6 Sol",
                  "value": 1,
                  "n": 2,
                  "highlight": false
                },
                {
                  "label": "GPT-6 Luna",
                  "value": 1,
                  "n": 2,
                  "highlight": false
                },
                {
                  "label": "GPT-6 Astra",
                  "value": 1,
                  "n": 2,
                  "highlight": false
                },
                {
                  "label": "GPT-5.5",
                  "value": 1,
                  "n": 2,
                  "highlight": false
                },
                {
                  "label": "Gemini 3.8 Flash",
                  "value": 1,
                  "n": 2,
                  "highlight": false
                },
                {
                  "label": "Gemini 3.5 Flash",
                  "value": 1,
                  "n": 2,
                  "highlight": false
                },
                {
                  "label": "Gemini 3.5 Flash Lite",
                  "value": 1,
                  "n": 2,
                  "highlight": false
                },
                {
                  "label": "Gemini 3.1 Pro Preview",
                  "value": 1,
                  "n": 2,
                  "highlight": false
                }
              ]
            }
          ],
          "note": "A calculation on prices reported by OpenRouter’s public API, snapshot 2026-10-06. n = providers with a standard-tier endpoint. 1x means every provider charges the same. Highlighted: a spread of 4x or more.",
          "sourceIds": [
            "openrouter-api-snapshot"
          ]
        },
        {
          "id": "gateway-markup-vs-first-party",
          "title": "OpenRouter markup over the first-party list price",
          "subtitle": "Per-token markup, and the markup after the Standard credit-purchase fee (calculation)",
          "kind": "grouped-bar",
          "unit": "percent",
          "yLabel": "Markup (%)",
          "series": [
            {
              "name": "Per-token markup (input)",
              "points": [
                {
                  "label": "Claude Haiku 4.5",
                  "value": 0
                },
                {
                  "label": "Claude Sonnet 5",
                  "value": 0
                },
                {
                  "label": "Claude Sonnet 5.5",
                  "value": 0
                },
                {
                  "label": "Claude Opus 4.8",
                  "value": 0
                },
                {
                  "label": "Claude Opus 5",
                  "value": 0
                },
                {
                  "label": "Claude Opus 5.5",
                  "value": 0
                },
                {
                  "label": "Claude Fable 5.1",
                  "value": 0
                },
                {
                  "label": "GPT-6 Luna",
                  "value": 0
                },
                {
                  "label": "Gemini 3.8 Flash",
                  "value": 0
                },
                {
                  "label": "Gemini 3.5 Flash",
                  "value": 0
                }
              ]
            },
            {
              "name": "Per-token markup (output)",
              "points": [
                {
                  "label": "Claude Haiku 4.5",
                  "value": 0
                },
                {
                  "label": "Claude Sonnet 5",
                  "value": 0
                },
                {
                  "label": "Claude Sonnet 5.5",
                  "value": 0
                },
                {
                  "label": "Claude Opus 4.8",
                  "value": 0
                },
                {
                  "label": "Claude Opus 5",
                  "value": 0
                },
                {
                  "label": "Claude Opus 5.5",
                  "value": 0
                },
                {
                  "label": "Claude Fable 5.1",
                  "value": 0
                },
                {
                  "label": "GPT-6 Luna",
                  "value": 0
                },
                {
                  "label": "Gemini 3.8 Flash",
                  "value": 0
                },
                {
                  "label": "Gemini 3.5 Flash",
                  "value": 0
                }
              ]
            },
            {
              "name": "With the 5.5% card credit fee (input)",
              "points": [
                {
                  "label": "Claude Haiku 4.5",
                  "value": 5.5
                },
                {
                  "label": "Claude Sonnet 5",
                  "value": 5.5
                },
                {
                  "label": "Claude Sonnet 5.5",
                  "value": 5.5
                },
                {
                  "label": "Claude Opus 4.8",
                  "value": 5.5
                },
                {
                  "label": "Claude Opus 5",
                  "value": 5.5
                },
                {
                  "label": "Claude Opus 5.5",
                  "value": 5.5
                },
                {
                  "label": "Claude Fable 5.1",
                  "value": 5.5
                },
                {
                  "label": "GPT-6 Luna",
                  "value": 5.5
                },
                {
                  "label": "Gemini 3.8 Flash",
                  "value": 5.5
                },
                {
                  "label": "Gemini 3.5 Flash",
                  "value": 5.5
                }
              ]
            }
          ],
          "note": "A calculation: OpenRouter list price ÷ first-party list price − 1, then × (1 + 5.5%) for credits bought by card on the Standard plan (purchases large enough that the minimum fee does not apply). Fees as published on 2026-10-06; first-party prices from the vendors’ list prices.",
          "sourceIds": [
            "openrouter-api-snapshot",
            "openrouter-fees",
            "price-anthropic",
            "price-openai",
            "price-google"
          ]
        },
        {
          "id": "gateway-vs-direct-claude-haiku-4-5",
          "title": "Claude Haiku 4.5: OpenRouter vs Anthropic list price",
          "subtitle": "USD per million tokens, before any credit-purchase fee",
          "kind": "grouped-bar",
          "unit": "usd",
          "yLabel": "USD per million tokens",
          "series": [
            {
              "name": "Input",
              "points": [
                {
                  "label": "OpenRouter",
                  "value": 1
                },
                {
                  "label": "Anthropic (first-party list price)",
                  "value": 1
                }
              ]
            },
            {
              "name": "Output",
              "points": [
                {
                  "label": "OpenRouter",
                  "value": 5
                },
                {
                  "label": "Anthropic (first-party list price)",
                  "value": 5
                }
              ]
            }
          ],
          "note": "OpenRouter price reported by OpenRouter’s public API, snapshot 2026-10-06. Anthropic price from the vendor’s published list price as recorded in the product price table. OpenRouter charges a 5.5% fee when credits are bought, not per request; the markup chart shows the price with that fee.",
          "factContext": "Claude Haiku 4.5 · list price, snapshot 2026-10-06",
          "sourceIds": [
            "openrouter-api-snapshot",
            "price-anthropic"
          ]
        },
        {
          "id": "gateway-vs-direct-claude-sonnet-5",
          "title": "Claude Sonnet 5: OpenRouter vs Anthropic list price",
          "subtitle": "USD per million tokens, before any credit-purchase fee",
          "kind": "grouped-bar",
          "unit": "usd",
          "yLabel": "USD per million tokens",
          "series": [
            {
              "name": "Input",
              "points": [
                {
                  "label": "OpenRouter",
                  "value": 2
                },
                {
                  "label": "Anthropic (first-party list price)",
                  "value": 2
                }
              ]
            },
            {
              "name": "Output",
              "points": [
                {
                  "label": "OpenRouter",
                  "value": 10
                },
                {
                  "label": "Anthropic (first-party list price)",
                  "value": 10
                }
              ]
            }
          ],
          "note": "OpenRouter price reported by OpenRouter’s public API, snapshot 2026-10-06. Anthropic price from the vendor’s published list price as recorded in the product price table. OpenRouter charges a 5.5% fee when credits are bought, not per request; the markup chart shows the price with that fee.",
          "factContext": "Claude Sonnet 5 · list price, snapshot 2026-10-06",
          "sourceIds": [
            "openrouter-api-snapshot",
            "price-anthropic"
          ]
        },
        {
          "id": "gateway-vs-direct-claude-sonnet-5-5",
          "title": "Claude Sonnet 5.5: OpenRouter vs Anthropic list price",
          "subtitle": "USD per million tokens, before any credit-purchase fee",
          "kind": "grouped-bar",
          "unit": "usd",
          "yLabel": "USD per million tokens",
          "series": [
            {
              "name": "Input",
              "points": [
                {
                  "label": "OpenRouter",
                  "value": 2
                },
                {
                  "label": "Anthropic (first-party list price)",
                  "value": 2
                }
              ]
            },
            {
              "name": "Output",
              "points": [
                {
                  "label": "OpenRouter",
                  "value": 10
                },
                {
                  "label": "Anthropic (first-party list price)",
                  "value": 10
                }
              ]
            }
          ],
          "note": "OpenRouter price reported by OpenRouter’s public API, snapshot 2026-10-06. Anthropic price from the vendor’s published list price as recorded in the product price table. OpenRouter charges a 5.5% fee when credits are bought, not per request; the markup chart shows the price with that fee.",
          "factContext": "Claude Sonnet 5.5 · list price, snapshot 2026-10-06",
          "sourceIds": [
            "openrouter-api-snapshot",
            "price-anthropic"
          ]
        },
        {
          "id": "gateway-vs-direct-claude-opus-4-8",
          "title": "Claude Opus 4.8: OpenRouter vs Anthropic list price",
          "subtitle": "USD per million tokens, before any credit-purchase fee",
          "kind": "grouped-bar",
          "unit": "usd",
          "yLabel": "USD per million tokens",
          "series": [
            {
              "name": "Input",
              "points": [
                {
                  "label": "OpenRouter",
                  "value": 5
                },
                {
                  "label": "Anthropic (first-party list price)",
                  "value": 5
                }
              ]
            },
            {
              "name": "Output",
              "points": [
                {
                  "label": "OpenRouter",
                  "value": 25
                },
                {
                  "label": "Anthropic (first-party list price)",
                  "value": 25
                }
              ]
            }
          ],
          "note": "OpenRouter price reported by OpenRouter’s public API, snapshot 2026-10-06. Anthropic price from the vendor’s published list price as recorded in the product price table. OpenRouter charges a 5.5% fee when credits are bought, not per request; the markup chart shows the price with that fee.",
          "factContext": "Claude Opus 4.8 · list price, snapshot 2026-10-06",
          "sourceIds": [
            "openrouter-api-snapshot",
            "price-anthropic"
          ]
        },
        {
          "id": "gateway-vs-direct-claude-opus-5",
          "title": "Claude Opus 5: OpenRouter vs Anthropic list price",
          "subtitle": "USD per million tokens, before any credit-purchase fee",
          "kind": "grouped-bar",
          "unit": "usd",
          "yLabel": "USD per million tokens",
          "series": [
            {
              "name": "Input",
              "points": [
                {
                  "label": "OpenRouter",
                  "value": 5
                },
                {
                  "label": "Anthropic (first-party list price)",
                  "value": 5
                }
              ]
            },
            {
              "name": "Output",
              "points": [
                {
                  "label": "OpenRouter",
                  "value": 25
                },
                {
                  "label": "Anthropic (first-party list price)",
                  "value": 25
                }
              ]
            }
          ],
          "note": "OpenRouter price reported by OpenRouter’s public API, snapshot 2026-10-06. Anthropic price from the vendor’s published list price as recorded in the product price table. OpenRouter charges a 5.5% fee when credits are bought, not per request; the markup chart shows the price with that fee.",
          "factContext": "Claude Opus 5 · list price, snapshot 2026-10-06",
          "sourceIds": [
            "openrouter-api-snapshot",
            "price-anthropic"
          ]
        },
        {
          "id": "gateway-vs-direct-claude-opus-5-5",
          "title": "Claude Opus 5.5: OpenRouter vs Anthropic list price",
          "subtitle": "USD per million tokens, before any credit-purchase fee",
          "kind": "grouped-bar",
          "unit": "usd",
          "yLabel": "USD per million tokens",
          "series": [
            {
              "name": "Input",
              "points": [
                {
                  "label": "OpenRouter",
                  "value": 4
                },
                {
                  "label": "Anthropic (first-party list price)",
                  "value": 4
                }
              ]
            },
            {
              "name": "Output",
              "points": [
                {
                  "label": "OpenRouter",
                  "value": 20
                },
                {
                  "label": "Anthropic (first-party list price)",
                  "value": 20
                }
              ]
            }
          ],
          "note": "OpenRouter price reported by OpenRouter’s public API, snapshot 2026-10-06. Anthropic price from the vendor’s published list price as recorded in the product price table. OpenRouter charges a 5.5% fee when credits are bought, not per request; the markup chart shows the price with that fee.",
          "factContext": "Claude Opus 5.5 · list price, snapshot 2026-10-06",
          "sourceIds": [
            "openrouter-api-snapshot",
            "price-anthropic"
          ]
        },
        {
          "id": "gateway-vs-direct-claude-fable-5-1",
          "title": "Claude Fable 5.1: OpenRouter vs Anthropic list price",
          "subtitle": "USD per million tokens, before any credit-purchase fee",
          "kind": "grouped-bar",
          "unit": "usd",
          "yLabel": "USD per million tokens",
          "series": [
            {
              "name": "Input",
              "points": [
                {
                  "label": "OpenRouter",
                  "value": 10
                },
                {
                  "label": "Anthropic (first-party list price)",
                  "value": 10
                }
              ]
            },
            {
              "name": "Output",
              "points": [
                {
                  "label": "OpenRouter",
                  "value": 50
                },
                {
                  "label": "Anthropic (first-party list price)",
                  "value": 50
                }
              ]
            }
          ],
          "note": "OpenRouter price reported by OpenRouter’s public API, snapshot 2026-10-06. Anthropic price from the vendor’s published list price as recorded in the product price table. OpenRouter charges a 5.5% fee when credits are bought, not per request; the markup chart shows the price with that fee.",
          "factContext": "Claude Fable 5.1 · list price, snapshot 2026-10-06",
          "sourceIds": [
            "openrouter-api-snapshot",
            "price-anthropic"
          ]
        },
        {
          "id": "gateway-vs-direct-gpt-6-luna",
          "title": "GPT-6 Luna: OpenRouter vs OpenAI list price",
          "subtitle": "USD per million tokens, before any credit-purchase fee",
          "kind": "grouped-bar",
          "unit": "usd",
          "yLabel": "USD per million tokens",
          "series": [
            {
              "name": "Input",
              "points": [
                {
                  "label": "OpenRouter",
                  "value": 0.1
                },
                {
                  "label": "OpenAI (first-party list price)",
                  "value": 0.1
                }
              ]
            },
            {
              "name": "Output",
              "points": [
                {
                  "label": "OpenRouter",
                  "value": 0.5
                },
                {
                  "label": "OpenAI (first-party list price)",
                  "value": 0.5
                }
              ]
            }
          ],
          "note": "OpenRouter price reported by OpenRouter’s public API, snapshot 2026-10-06. OpenAI price from the vendor’s published list price as recorded in the product price table. OpenRouter charges a 5.5% fee when credits are bought, not per request; the markup chart shows the price with that fee.",
          "factContext": "GPT-6 Luna · list price, snapshot 2026-10-06",
          "sourceIds": [
            "openrouter-api-snapshot",
            "price-openai"
          ]
        },
        {
          "id": "gateway-vs-direct-gemini-3-8-flash",
          "title": "Gemini 3.8 Flash: OpenRouter vs Google list price",
          "subtitle": "USD per million tokens, before any credit-purchase fee",
          "kind": "grouped-bar",
          "unit": "usd",
          "yLabel": "USD per million tokens",
          "series": [
            {
              "name": "Input",
              "points": [
                {
                  "label": "OpenRouter",
                  "value": 0.75
                },
                {
                  "label": "Google AI Studio (first-party list price)",
                  "value": 0.75
                }
              ]
            },
            {
              "name": "Output",
              "points": [
                {
                  "label": "OpenRouter",
                  "value": 3.75
                },
                {
                  "label": "Google AI Studio (first-party list price)",
                  "value": 3.75
                }
              ]
            }
          ],
          "note": "OpenRouter price reported by OpenRouter’s public API, snapshot 2026-10-06. Google price from the vendor’s published list price as recorded in the product price table. OpenRouter charges a 5.5% fee when credits are bought, not per request; the markup chart shows the price with that fee.",
          "factContext": "Gemini 3.8 Flash · list price, snapshot 2026-10-06",
          "sourceIds": [
            "openrouter-api-snapshot",
            "price-google"
          ]
        },
        {
          "id": "gateway-vs-direct-gemini-3-5-flash",
          "title": "Gemini 3.5 Flash: OpenRouter vs Google list price",
          "subtitle": "USD per million tokens, before any credit-purchase fee",
          "kind": "grouped-bar",
          "unit": "usd",
          "yLabel": "USD per million tokens",
          "series": [
            {
              "name": "Input",
              "points": [
                {
                  "label": "OpenRouter",
                  "value": 1.5
                },
                {
                  "label": "Google AI Studio (first-party list price)",
                  "value": 1.5
                }
              ]
            },
            {
              "name": "Output",
              "points": [
                {
                  "label": "OpenRouter",
                  "value": 9
                },
                {
                  "label": "Google AI Studio (first-party list price)",
                  "value": 9
                }
              ]
            }
          ],
          "note": "OpenRouter price reported by OpenRouter’s public API, snapshot 2026-10-06. Google price from the vendor’s published list price as recorded in the product price table. OpenRouter charges a 5.5% fee when credits are bought, not per request; the markup chart shows the price with that fee.",
          "factContext": "Gemini 3.5 Flash · list price, snapshot 2026-10-06",
          "sourceIds": [
            "openrouter-api-snapshot",
            "price-google"
          ]
        },
        {
          "id": "provider-prices-claude-haiku-4-5",
          "title": "Claude Haiku 4.5: price per million tokens by provider",
          "subtitle": "Standard tier, one bar per provider (its cheapest standard endpoint); reported by OpenRouter’s public API, snapshot 2026-10-06",
          "kind": "grouped-bar",
          "unit": "usd",
          "yLabel": "USD per million tokens",
          "series": [
            {
              "name": "Input",
              "points": [
                {
                  "label": "Amazon Bedrock",
                  "value": 1
                },
                {
                  "label": "Anthropic",
                  "value": 1
                },
                {
                  "label": "Azure",
                  "value": 1
                },
                {
                  "label": "Google Vertex",
                  "value": 1
                }
              ]
            },
            {
              "name": "Output",
              "points": [
                {
                  "label": "Amazon Bedrock",
                  "value": 5
                },
                {
                  "label": "Anthropic",
                  "value": 5
                },
                {
                  "label": "Azure",
                  "value": 5
                },
                {
                  "label": "Google Vertex",
                  "value": 5
                }
              ]
            },
            {
              "name": "Cache read",
              "points": [
                {
                  "label": "Amazon Bedrock",
                  "value": 0.1
                },
                {
                  "label": "Anthropic",
                  "value": 0.1
                },
                {
                  "label": "Azure",
                  "value": 0.1
                },
                {
                  "label": "Google Vertex",
                  "value": 0.1
                }
              ]
            }
          ],
          "note": "Prices reported by OpenRouter’s public API, snapshot 2026-10-06; third-party-reported, not measured by Agent. Sorted from the lowest to the highest blended price (3 input : 1 output). A parenthesis names the quantization the provider reported; lower precision or a shorter context can explain a lower price, so check the endpoint table. Flex, priority, fast and regional endpoints are left out here because they are priced differently on purpose.",
          "factContext": "Claude Haiku 4.5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "sourceIds": [
            "openrouter-api-snapshot"
          ]
        },
        {
          "id": "provider-prices-claude-sonnet-5",
          "title": "Claude Sonnet 5: price per million tokens by provider",
          "subtitle": "Standard tier, one bar per provider (its cheapest standard endpoint); reported by OpenRouter’s public API, snapshot 2026-10-06",
          "kind": "grouped-bar",
          "unit": "usd",
          "yLabel": "USD per million tokens",
          "series": [
            {
              "name": "Input",
              "points": [
                {
                  "label": "Amazon Bedrock",
                  "value": 2
                },
                {
                  "label": "Anthropic",
                  "value": 2
                },
                {
                  "label": "Azure",
                  "value": 2
                },
                {
                  "label": "Claude Platform on AWS",
                  "value": 2
                },
                {
                  "label": "Google Vertex",
                  "value": 2
                }
              ]
            },
            {
              "name": "Output",
              "points": [
                {
                  "label": "Amazon Bedrock",
                  "value": 10
                },
                {
                  "label": "Anthropic",
                  "value": 10
                },
                {
                  "label": "Azure",
                  "value": 10
                },
                {
                  "label": "Claude Platform on AWS",
                  "value": 10
                },
                {
                  "label": "Google Vertex",
                  "value": 10
                }
              ]
            },
            {
              "name": "Cache read",
              "points": [
                {
                  "label": "Amazon Bedrock",
                  "value": 0.2
                },
                {
                  "label": "Anthropic",
                  "value": 0.2
                },
                {
                  "label": "Azure",
                  "value": 0.2
                },
                {
                  "label": "Claude Platform on AWS",
                  "value": 0.2
                },
                {
                  "label": "Google Vertex",
                  "value": 0.2
                }
              ]
            }
          ],
          "note": "Prices reported by OpenRouter’s public API, snapshot 2026-10-06; third-party-reported, not measured by Agent. Sorted from the lowest to the highest blended price (3 input : 1 output). A parenthesis names the quantization the provider reported; lower precision or a shorter context can explain a lower price, so check the endpoint table. Flex, priority, fast and regional endpoints are left out here because they are priced differently on purpose.",
          "factContext": "Claude Sonnet 5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "sourceIds": [
            "openrouter-api-snapshot"
          ]
        },
        {
          "id": "provider-prices-claude-sonnet-5-5",
          "title": "Claude Sonnet 5.5: price per million tokens by provider",
          "subtitle": "Standard tier, one bar per provider (its cheapest standard endpoint); reported by OpenRouter’s public API, snapshot 2026-10-06",
          "kind": "grouped-bar",
          "unit": "usd",
          "yLabel": "USD per million tokens",
          "series": [
            {
              "name": "Input",
              "points": [
                {
                  "label": "Amazon Bedrock",
                  "value": 2
                },
                {
                  "label": "Anthropic",
                  "value": 2
                },
                {
                  "label": "Azure",
                  "value": 2
                },
                {
                  "label": "Claude Platform on AWS",
                  "value": 2
                },
                {
                  "label": "Google Vertex",
                  "value": 2
                }
              ]
            },
            {
              "name": "Output",
              "points": [
                {
                  "label": "Amazon Bedrock",
                  "value": 10
                },
                {
                  "label": "Anthropic",
                  "value": 10
                },
                {
                  "label": "Azure",
                  "value": 10
                },
                {
                  "label": "Claude Platform on AWS",
                  "value": 10
                },
                {
                  "label": "Google Vertex",
                  "value": 10
                }
              ]
            },
            {
              "name": "Cache read",
              "points": [
                {
                  "label": "Amazon Bedrock",
                  "value": 0.2
                },
                {
                  "label": "Anthropic",
                  "value": 0.2
                },
                {
                  "label": "Azure",
                  "value": 0.2
                },
                {
                  "label": "Claude Platform on AWS",
                  "value": 0.2
                },
                {
                  "label": "Google Vertex",
                  "value": 0.2
                }
              ]
            }
          ],
          "note": "Prices reported by OpenRouter’s public API, snapshot 2026-10-06; third-party-reported, not measured by Agent. Sorted from the lowest to the highest blended price (3 input : 1 output). A parenthesis names the quantization the provider reported; lower precision or a shorter context can explain a lower price, so check the endpoint table. Flex, priority, fast and regional endpoints are left out here because they are priced differently on purpose.",
          "factContext": "Claude Sonnet 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "sourceIds": [
            "openrouter-api-snapshot"
          ]
        },
        {
          "id": "provider-prices-claude-opus-4-8",
          "title": "Claude Opus 4.8: price per million tokens by provider",
          "subtitle": "Standard tier, one bar per provider (its cheapest standard endpoint); reported by OpenRouter’s public API, snapshot 2026-10-06",
          "kind": "grouped-bar",
          "unit": "usd",
          "yLabel": "USD per million tokens",
          "series": [
            {
              "name": "Input",
              "points": [
                {
                  "label": "Amazon Bedrock",
                  "value": 5
                },
                {
                  "label": "Anthropic",
                  "value": 5
                },
                {
                  "label": "Azure",
                  "value": 5
                },
                {
                  "label": "Claude Platform on AWS",
                  "value": 5
                },
                {
                  "label": "Google Vertex",
                  "value": 5
                }
              ]
            },
            {
              "name": "Output",
              "points": [
                {
                  "label": "Amazon Bedrock",
                  "value": 25
                },
                {
                  "label": "Anthropic",
                  "value": 25
                },
                {
                  "label": "Azure",
                  "value": 25
                },
                {
                  "label": "Claude Platform on AWS",
                  "value": 25
                },
                {
                  "label": "Google Vertex",
                  "value": 25
                }
              ]
            },
            {
              "name": "Cache read",
              "points": [
                {
                  "label": "Amazon Bedrock",
                  "value": 0.5
                },
                {
                  "label": "Anthropic",
                  "value": 0.5
                },
                {
                  "label": "Azure",
                  "value": 0.5
                },
                {
                  "label": "Claude Platform on AWS",
                  "value": 0.5
                },
                {
                  "label": "Google Vertex",
                  "value": 0.5
                }
              ]
            }
          ],
          "note": "Prices reported by OpenRouter’s public API, snapshot 2026-10-06; third-party-reported, not measured by Agent. Sorted from the lowest to the highest blended price (3 input : 1 output). A parenthesis names the quantization the provider reported; lower precision or a shorter context can explain a lower price, so check the endpoint table. Flex, priority, fast and regional endpoints are left out here because they are priced differently on purpose.",
          "factContext": "Claude Opus 4.8 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "sourceIds": [
            "openrouter-api-snapshot"
          ]
        },
        {
          "id": "provider-prices-claude-opus-5",
          "title": "Claude Opus 5: price per million tokens by provider",
          "subtitle": "Standard tier, one bar per provider (its cheapest standard endpoint); reported by OpenRouter’s public API, snapshot 2026-10-06",
          "kind": "grouped-bar",
          "unit": "usd",
          "yLabel": "USD per million tokens",
          "series": [
            {
              "name": "Input",
              "points": [
                {
                  "label": "Amazon Bedrock",
                  "value": 5
                },
                {
                  "label": "Anthropic",
                  "value": 5
                },
                {
                  "label": "Azure",
                  "value": 5
                },
                {
                  "label": "Claude Platform on AWS",
                  "value": 5
                },
                {
                  "label": "Google Vertex",
                  "value": 5
                }
              ]
            },
            {
              "name": "Output",
              "points": [
                {
                  "label": "Amazon Bedrock",
                  "value": 25
                },
                {
                  "label": "Anthropic",
                  "value": 25
                },
                {
                  "label": "Azure",
                  "value": 25
                },
                {
                  "label": "Claude Platform on AWS",
                  "value": 25
                },
                {
                  "label": "Google Vertex",
                  "value": 25
                }
              ]
            },
            {
              "name": "Cache read",
              "points": [
                {
                  "label": "Amazon Bedrock",
                  "value": 0.5
                },
                {
                  "label": "Anthropic",
                  "value": 0.5
                },
                {
                  "label": "Azure",
                  "value": 0.5
                },
                {
                  "label": "Claude Platform on AWS",
                  "value": 0.5
                },
                {
                  "label": "Google Vertex",
                  "value": 0.5
                }
              ]
            }
          ],
          "note": "Prices reported by OpenRouter’s public API, snapshot 2026-10-06; third-party-reported, not measured by Agent. Sorted from the lowest to the highest blended price (3 input : 1 output). A parenthesis names the quantization the provider reported; lower precision or a shorter context can explain a lower price, so check the endpoint table. Flex, priority, fast and regional endpoints are left out here because they are priced differently on purpose.",
          "factContext": "Claude Opus 5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "sourceIds": [
            "openrouter-api-snapshot"
          ]
        },
        {
          "id": "provider-prices-claude-opus-5-5",
          "title": "Claude Opus 5.5: price per million tokens by provider",
          "subtitle": "Standard tier, one bar per provider (its cheapest standard endpoint); reported by OpenRouter’s public API, snapshot 2026-10-06",
          "kind": "grouped-bar",
          "unit": "usd",
          "yLabel": "USD per million tokens",
          "series": [
            {
              "name": "Input",
              "points": [
                {
                  "label": "Amazon Bedrock",
                  "value": 4
                },
                {
                  "label": "Anthropic",
                  "value": 4
                },
                {
                  "label": "Azure",
                  "value": 4
                },
                {
                  "label": "Claude Platform on AWS",
                  "value": 4
                },
                {
                  "label": "Google Vertex",
                  "value": 4
                }
              ]
            },
            {
              "name": "Output",
              "points": [
                {
                  "label": "Amazon Bedrock",
                  "value": 20
                },
                {
                  "label": "Anthropic",
                  "value": 20
                },
                {
                  "label": "Azure",
                  "value": 20
                },
                {
                  "label": "Claude Platform on AWS",
                  "value": 20
                },
                {
                  "label": "Google Vertex",
                  "value": 20
                }
              ]
            },
            {
              "name": "Cache read",
              "points": [
                {
                  "label": "Amazon Bedrock",
                  "value": 0.2
                },
                {
                  "label": "Anthropic",
                  "value": 0.2
                },
                {
                  "label": "Azure",
                  "value": 0.2
                },
                {
                  "label": "Claude Platform on AWS",
                  "value": 0.2
                },
                {
                  "label": "Google Vertex",
                  "value": 0.2
                }
              ]
            }
          ],
          "note": "Prices reported by OpenRouter’s public API, snapshot 2026-10-06; third-party-reported, not measured by Agent. Sorted from the lowest to the highest blended price (3 input : 1 output). A parenthesis names the quantization the provider reported; lower precision or a shorter context can explain a lower price, so check the endpoint table. Flex, priority, fast and regional endpoints are left out here because they are priced differently on purpose.",
          "factContext": "Claude Opus 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "sourceIds": [
            "openrouter-api-snapshot"
          ]
        },
        {
          "id": "provider-prices-claude-fable-5-1",
          "title": "Claude Fable 5.1: price per million tokens by provider",
          "subtitle": "Standard tier, one bar per provider (its cheapest standard endpoint); reported by OpenRouter’s public API, snapshot 2026-10-06",
          "kind": "grouped-bar",
          "unit": "usd",
          "yLabel": "USD per million tokens",
          "series": [
            {
              "name": "Input",
              "points": [
                {
                  "label": "Amazon Bedrock",
                  "value": 10
                },
                {
                  "label": "Anthropic",
                  "value": 10
                },
                {
                  "label": "Azure",
                  "value": 10
                },
                {
                  "label": "Google Vertex",
                  "value": 10
                }
              ]
            },
            {
              "name": "Output",
              "points": [
                {
                  "label": "Amazon Bedrock",
                  "value": 50
                },
                {
                  "label": "Anthropic",
                  "value": 50
                },
                {
                  "label": "Azure",
                  "value": 50
                },
                {
                  "label": "Google Vertex",
                  "value": 50
                }
              ]
            },
            {
              "name": "Cache read",
              "points": [
                {
                  "label": "Amazon Bedrock",
                  "value": 0.25
                },
                {
                  "label": "Anthropic",
                  "value": 0.25
                },
                {
                  "label": "Azure",
                  "value": 0.25
                },
                {
                  "label": "Google Vertex",
                  "value": 0.25
                }
              ]
            }
          ],
          "note": "Prices reported by OpenRouter’s public API, snapshot 2026-10-06; third-party-reported, not measured by Agent. Sorted from the lowest to the highest blended price (3 input : 1 output). A parenthesis names the quantization the provider reported; lower precision or a shorter context can explain a lower price, so check the endpoint table. Flex, priority, fast and regional endpoints are left out here because they are priced differently on purpose.",
          "factContext": "Claude Fable 5.1 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "sourceIds": [
            "openrouter-api-snapshot"
          ]
        },
        {
          "id": "provider-prices-gpt-6-sol",
          "title": "GPT-6 Sol: price per million tokens by provider",
          "subtitle": "Standard tier, one bar per provider (its cheapest standard endpoint); reported by OpenRouter’s public API, snapshot 2026-10-06",
          "kind": "grouped-bar",
          "unit": "usd",
          "yLabel": "USD per million tokens",
          "series": [
            {
              "name": "Input",
              "points": [
                {
                  "label": "Azure",
                  "value": 2
                },
                {
                  "label": "OpenAI",
                  "value": 2
                }
              ]
            },
            {
              "name": "Output",
              "points": [
                {
                  "label": "Azure",
                  "value": 10
                },
                {
                  "label": "OpenAI",
                  "value": 10
                }
              ]
            },
            {
              "name": "Cache read",
              "points": [
                {
                  "label": "Azure",
                  "value": 0.2
                },
                {
                  "label": "OpenAI",
                  "value": 0.2
                }
              ]
            }
          ],
          "note": "Prices reported by OpenRouter’s public API, snapshot 2026-10-06; third-party-reported, not measured by Agent. Sorted from the lowest to the highest blended price (3 input : 1 output). A parenthesis names the quantization the provider reported; lower precision or a shorter context can explain a lower price, so check the endpoint table. Flex, priority, fast and regional endpoints are left out here because they are priced differently on purpose.",
          "factContext": "GPT-6 Sol · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "sourceIds": [
            "openrouter-api-snapshot"
          ]
        },
        {
          "id": "provider-prices-gpt-6-luna",
          "title": "GPT-6 Luna: price per million tokens by provider",
          "subtitle": "Standard tier, one bar per provider (its cheapest standard endpoint); reported by OpenRouter’s public API, snapshot 2026-10-06",
          "kind": "grouped-bar",
          "unit": "usd",
          "yLabel": "USD per million tokens",
          "series": [
            {
              "name": "Input",
              "points": [
                {
                  "label": "Azure",
                  "value": 0.1
                },
                {
                  "label": "OpenAI",
                  "value": 0.1
                }
              ]
            },
            {
              "name": "Output",
              "points": [
                {
                  "label": "Azure",
                  "value": 0.5
                },
                {
                  "label": "OpenAI",
                  "value": 0.5
                }
              ]
            },
            {
              "name": "Cache read",
              "points": [
                {
                  "label": "Azure",
                  "value": 0.01
                },
                {
                  "label": "OpenAI",
                  "value": 0.01
                }
              ]
            }
          ],
          "note": "Prices reported by OpenRouter’s public API, snapshot 2026-10-06; third-party-reported, not measured by Agent. Sorted from the lowest to the highest blended price (3 input : 1 output). A parenthesis names the quantization the provider reported; lower precision or a shorter context can explain a lower price, so check the endpoint table. Flex, priority, fast and regional endpoints are left out here because they are priced differently on purpose.",
          "factContext": "GPT-6 Luna · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "sourceIds": [
            "openrouter-api-snapshot"
          ]
        },
        {
          "id": "provider-prices-gpt-6-astra",
          "title": "GPT-6 Astra: price per million tokens by provider",
          "subtitle": "Standard tier, one bar per provider (its cheapest standard endpoint); reported by OpenRouter’s public API, snapshot 2026-10-06",
          "kind": "grouped-bar",
          "unit": "usd",
          "yLabel": "USD per million tokens",
          "series": [
            {
              "name": "Input",
              "points": [
                {
                  "label": "Azure",
                  "value": 10
                },
                {
                  "label": "OpenAI",
                  "value": 10
                }
              ]
            },
            {
              "name": "Output",
              "points": [
                {
                  "label": "Azure",
                  "value": 50
                },
                {
                  "label": "OpenAI",
                  "value": 50
                }
              ]
            },
            {
              "name": "Cache read",
              "points": [
                {
                  "label": "Azure",
                  "value": 1
                },
                {
                  "label": "OpenAI",
                  "value": 1
                }
              ]
            }
          ],
          "note": "Prices reported by OpenRouter’s public API, snapshot 2026-10-06; third-party-reported, not measured by Agent. Sorted from the lowest to the highest blended price (3 input : 1 output). A parenthesis names the quantization the provider reported; lower precision or a shorter context can explain a lower price, so check the endpoint table. Flex, priority, fast and regional endpoints are left out here because they are priced differently on purpose.",
          "factContext": "GPT-6 Astra · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "sourceIds": [
            "openrouter-api-snapshot"
          ]
        },
        {
          "id": "provider-prices-gpt-5-5",
          "title": "GPT-5.5: price per million tokens by provider",
          "subtitle": "Standard tier, one bar per provider (its cheapest standard endpoint); reported by OpenRouter’s public API, snapshot 2026-10-06",
          "kind": "grouped-bar",
          "unit": "usd",
          "yLabel": "USD per million tokens",
          "series": [
            {
              "name": "Input",
              "points": [
                {
                  "label": "Azure",
                  "value": 5
                },
                {
                  "label": "OpenAI",
                  "value": 5
                }
              ]
            },
            {
              "name": "Output",
              "points": [
                {
                  "label": "Azure",
                  "value": 30
                },
                {
                  "label": "OpenAI",
                  "value": 30
                }
              ]
            },
            {
              "name": "Cache read",
              "points": [
                {
                  "label": "Azure",
                  "value": 0.5
                },
                {
                  "label": "OpenAI",
                  "value": 0.5
                }
              ]
            }
          ],
          "note": "Prices reported by OpenRouter’s public API, snapshot 2026-10-06; third-party-reported, not measured by Agent. Sorted from the lowest to the highest blended price (3 input : 1 output). A parenthesis names the quantization the provider reported; lower precision or a shorter context can explain a lower price, so check the endpoint table. Flex, priority, fast and regional endpoints are left out here because they are priced differently on purpose.",
          "factContext": "GPT-5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "sourceIds": [
            "openrouter-api-snapshot"
          ]
        },
        {
          "id": "provider-prices-gpt-oss-120b",
          "title": "gpt-oss-120b: price per million tokens by provider",
          "subtitle": "Standard tier, one bar per provider (its cheapest standard endpoint); reported by OpenRouter’s public API, snapshot 2026-10-06",
          "kind": "grouped-bar",
          "unit": "usd",
          "yLabel": "USD per million tokens",
          "series": [
            {
              "name": "Input",
              "points": [
                {
                  "label": "CoreWeave (fp4)",
                  "value": 0.03
                },
                {
                  "label": "DekaLLM (bf16)",
                  "value": 0.03
                },
                {
                  "label": "DeepInfra (bf16)",
                  "value": 0.037
                },
                {
                  "label": "AkashML (bf16)",
                  "value": 0.037
                },
                {
                  "label": "Mancer 2 (fp8)",
                  "value": 0.045
                },
                {
                  "label": "Crusoe (bf16)",
                  "value": 0.05
                },
                {
                  "label": "Novita (fp4)",
                  "value": 0.05
                },
                {
                  "label": "DigitalOcean",
                  "value": 0.06
                },
                {
                  "label": "Google Vertex",
                  "value": 0.09
                },
                {
                  "label": "BaseTen (fp4)",
                  "value": 0.1
                },
                {
                  "label": "Amazon Bedrock",
                  "value": 0.15
                },
                {
                  "label": "Groq",
                  "value": 0.15
                },
                {
                  "label": "Nebius (fp4)",
                  "value": 0.15
                },
                {
                  "label": "Phala",
                  "value": 0.15
                },
                {
                  "label": "SiliconFlow (fp8)",
                  "value": 0.15
                },
                {
                  "label": "Together",
                  "value": 0.15
                },
                {
                  "label": "Parasail (fp4)",
                  "value": 0.1
                },
                {
                  "label": "Mara",
                  "value": 0.15
                },
                {
                  "label": "SambaNova",
                  "value": 0.14
                },
                {
                  "label": "Cerebras (fp16)",
                  "value": 0.35
                }
              ]
            },
            {
              "name": "Output",
              "points": [
                {
                  "label": "CoreWeave (fp4)",
                  "value": 0.17
                },
                {
                  "label": "DekaLLM (bf16)",
                  "value": 0.18
                },
                {
                  "label": "DeepInfra (bf16)",
                  "value": 0.17
                },
                {
                  "label": "AkashML (bf16)",
                  "value": 0.187
                },
                {
                  "label": "Mancer 2 (fp8)",
                  "value": 0.25
                },
                {
                  "label": "Crusoe (bf16)",
                  "value": 0.25
                },
                {
                  "label": "Novita (fp4)",
                  "value": 0.25
                },
                {
                  "label": "DigitalOcean",
                  "value": 0.42
                },
                {
                  "label": "Google Vertex",
                  "value": 0.36
                },
                {
                  "label": "BaseTen (fp4)",
                  "value": 0.5
                },
                {
                  "label": "Amazon Bedrock",
                  "value": 0.6
                },
                {
                  "label": "Groq",
                  "value": 0.6
                },
                {
                  "label": "Nebius (fp4)",
                  "value": 0.6
                },
                {
                  "label": "Phala",
                  "value": 0.6
                },
                {
                  "label": "SiliconFlow (fp8)",
                  "value": 0.6
                },
                {
                  "label": "Together",
                  "value": 0.6
                },
                {
                  "label": "Parasail (fp4)",
                  "value": 0.75
                },
                {
                  "label": "Mara",
                  "value": 0.75
                },
                {
                  "label": "SambaNova",
                  "value": 0.95
                },
                {
                  "label": "Cerebras (fp16)",
                  "value": 0.75
                }
              ]
            },
            {
              "name": "Cache read",
              "points": [
                {
                  "label": "CoreWeave (fp4)",
                  "value": 0.03
                },
                {
                  "label": "DekaLLM (bf16)",
                  "value": 0.03
                },
                {
                  "label": "AkashML (bf16)",
                  "value": 0.037
                },
                {
                  "label": "Crusoe (bf16)",
                  "value": 0.05
                },
                {
                  "label": "DigitalOcean",
                  "value": 0.012
                },
                {
                  "label": "BaseTen (fp4)",
                  "value": 0.1
                },
                {
                  "label": "Groq",
                  "value": 0.075
                },
                {
                  "label": "SiliconFlow (fp8)",
                  "value": 0.075
                },
                {
                  "label": "Parasail (fp4)",
                  "value": 0.055
                },
                {
                  "label": "Cerebras (fp16)",
                  "value": 0.35
                }
              ]
            }
          ],
          "note": "Prices reported by OpenRouter’s public API, snapshot 2026-10-06; third-party-reported, not measured by Agent. Sorted from the lowest to the highest blended price (3 input : 1 output). A parenthesis names the quantization the provider reported; lower precision or a shorter context can explain a lower price, so check the endpoint table. Flex, priority, fast and regional endpoints are left out here because they are priced differently on purpose.",
          "factContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "sourceIds": [
            "openrouter-api-snapshot"
          ]
        },
        {
          "id": "provider-prices-gemini-3-8-flash",
          "title": "Gemini 3.8 Flash: price per million tokens by provider",
          "subtitle": "Standard tier, one bar per provider (its cheapest standard endpoint); reported by OpenRouter’s public API, snapshot 2026-10-06",
          "kind": "grouped-bar",
          "unit": "usd",
          "yLabel": "USD per million tokens",
          "series": [
            {
              "name": "Input",
              "points": [
                {
                  "label": "Google AI Studio",
                  "value": 0.75
                },
                {
                  "label": "Google Vertex",
                  "value": 0.75
                }
              ]
            },
            {
              "name": "Output",
              "points": [
                {
                  "label": "Google AI Studio",
                  "value": 3.75
                },
                {
                  "label": "Google Vertex",
                  "value": 3.75
                }
              ]
            },
            {
              "name": "Cache read",
              "points": [
                {
                  "label": "Google AI Studio",
                  "value": 0.075
                },
                {
                  "label": "Google Vertex",
                  "value": 0.075
                }
              ]
            }
          ],
          "note": "Prices reported by OpenRouter’s public API, snapshot 2026-10-06; third-party-reported, not measured by Agent. Sorted from the lowest to the highest blended price (3 input : 1 output). A parenthesis names the quantization the provider reported; lower precision or a shorter context can explain a lower price, so check the endpoint table. Flex, priority, fast and regional endpoints are left out here because they are priced differently on purpose.",
          "factContext": "Gemini 3.8 Flash · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "sourceIds": [
            "openrouter-api-snapshot"
          ]
        },
        {
          "id": "provider-prices-gemini-3-5-flash",
          "title": "Gemini 3.5 Flash: price per million tokens by provider",
          "subtitle": "Standard tier, one bar per provider (its cheapest standard endpoint); reported by OpenRouter’s public API, snapshot 2026-10-06",
          "kind": "grouped-bar",
          "unit": "usd",
          "yLabel": "USD per million tokens",
          "series": [
            {
              "name": "Input",
              "points": [
                {
                  "label": "Google AI Studio",
                  "value": 1.5
                },
                {
                  "label": "Google Vertex",
                  "value": 1.5
                }
              ]
            },
            {
              "name": "Output",
              "points": [
                {
                  "label": "Google AI Studio",
                  "value": 9
                },
                {
                  "label": "Google Vertex",
                  "value": 9
                }
              ]
            },
            {
              "name": "Cache read",
              "points": [
                {
                  "label": "Google AI Studio",
                  "value": 0.15
                },
                {
                  "label": "Google Vertex",
                  "value": 0.15
                }
              ]
            }
          ],
          "note": "Prices reported by OpenRouter’s public API, snapshot 2026-10-06; third-party-reported, not measured by Agent. Sorted from the lowest to the highest blended price (3 input : 1 output). A parenthesis names the quantization the provider reported; lower precision or a shorter context can explain a lower price, so check the endpoint table. Flex, priority, fast and regional endpoints are left out here because they are priced differently on purpose.",
          "factContext": "Gemini 3.5 Flash · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "sourceIds": [
            "openrouter-api-snapshot"
          ]
        },
        {
          "id": "provider-prices-gemini-3-5-flash-lite",
          "title": "Gemini 3.5 Flash Lite: price per million tokens by provider",
          "subtitle": "Standard tier, one bar per provider (its cheapest standard endpoint); reported by OpenRouter’s public API, snapshot 2026-10-06",
          "kind": "grouped-bar",
          "unit": "usd",
          "yLabel": "USD per million tokens",
          "series": [
            {
              "name": "Input",
              "points": [
                {
                  "label": "Google AI Studio",
                  "value": 0.3
                },
                {
                  "label": "Google Vertex",
                  "value": 0.3
                }
              ]
            },
            {
              "name": "Output",
              "points": [
                {
                  "label": "Google AI Studio",
                  "value": 2.5
                },
                {
                  "label": "Google Vertex",
                  "value": 2.5
                }
              ]
            },
            {
              "name": "Cache read",
              "points": [
                {
                  "label": "Google AI Studio",
                  "value": 0.03
                },
                {
                  "label": "Google Vertex",
                  "value": 0.03
                }
              ]
            }
          ],
          "note": "Prices reported by OpenRouter’s public API, snapshot 2026-10-06; third-party-reported, not measured by Agent. Sorted from the lowest to the highest blended price (3 input : 1 output). A parenthesis names the quantization the provider reported; lower precision or a shorter context can explain a lower price, so check the endpoint table. Flex, priority, fast and regional endpoints are left out here because they are priced differently on purpose.",
          "factContext": "Gemini 3.5 Flash Lite · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "sourceIds": [
            "openrouter-api-snapshot"
          ]
        },
        {
          "id": "provider-prices-gemini-3-1-pro-preview",
          "title": "Gemini 3.1 Pro Preview: price per million tokens by provider",
          "subtitle": "Standard tier, one bar per provider (its cheapest standard endpoint); reported by OpenRouter’s public API, snapshot 2026-10-06",
          "kind": "grouped-bar",
          "unit": "usd",
          "yLabel": "USD per million tokens",
          "series": [
            {
              "name": "Input",
              "points": [
                {
                  "label": "Google AI Studio",
                  "value": 2
                },
                {
                  "label": "Google Vertex",
                  "value": 2
                }
              ]
            },
            {
              "name": "Output",
              "points": [
                {
                  "label": "Google AI Studio",
                  "value": 12
                },
                {
                  "label": "Google Vertex",
                  "value": 12
                }
              ]
            },
            {
              "name": "Cache read",
              "points": [
                {
                  "label": "Google AI Studio",
                  "value": 0.2
                },
                {
                  "label": "Google Vertex",
                  "value": 0.2
                }
              ]
            }
          ],
          "note": "Prices reported by OpenRouter’s public API, snapshot 2026-10-06; third-party-reported, not measured by Agent. Sorted from the lowest to the highest blended price (3 input : 1 output). A parenthesis names the quantization the provider reported; lower precision or a shorter context can explain a lower price, so check the endpoint table. Flex, priority, fast and regional endpoints are left out here because they are priced differently on purpose.",
          "factContext": "Gemini 3.1 Pro Preview · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "sourceIds": [
            "openrouter-api-snapshot"
          ]
        },
        {
          "id": "provider-prices-llama-4-maverick",
          "title": "Llama 4 Maverick: price per million tokens by provider",
          "subtitle": "Standard tier, one bar per provider (its cheapest standard endpoint); reported by OpenRouter’s public API, snapshot 2026-10-06",
          "kind": "grouped-bar",
          "unit": "usd",
          "yLabel": "USD per million tokens",
          "series": [
            {
              "name": "Input",
              "points": [
                {
                  "label": "DigitalOcean",
                  "value": 0.1875
                },
                {
                  "label": "Novita (fp8)",
                  "value": 0.27
                },
                {
                  "label": "Parasail (fp8)",
                  "value": 0.35
                }
              ]
            },
            {
              "name": "Output",
              "points": [
                {
                  "label": "DigitalOcean",
                  "value": 0.6525
                },
                {
                  "label": "Novita (fp8)",
                  "value": 0.85
                },
                {
                  "label": "Parasail (fp8)",
                  "value": 1
                }
              ]
            }
          ],
          "note": "Prices reported by OpenRouter’s public API, snapshot 2026-10-06; third-party-reported, not measured by Agent. Sorted from the lowest to the highest blended price (3 input : 1 output). A parenthesis names the quantization the provider reported; lower precision or a shorter context can explain a lower price, so check the endpoint table. Flex, priority, fast and regional endpoints are left out here because they are priced differently on purpose.",
          "factContext": "Llama 4 Maverick · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "sourceIds": [
            "openrouter-api-snapshot"
          ]
        },
        {
          "id": "provider-prices-llama-3-3-70b-instruct",
          "title": "Llama 3.3 70B Instruct: price per million tokens by provider",
          "subtitle": "Standard tier, one bar per provider (its cheapest standard endpoint); reported by OpenRouter’s public API, snapshot 2026-10-06",
          "kind": "grouped-bar",
          "unit": "usd",
          "yLabel": "USD per million tokens",
          "series": [
            {
              "name": "Input",
              "points": [
                {
                  "label": "DeepInfra (fp8)",
                  "value": 0.1
                },
                {
                  "label": "Novita (bf16)",
                  "value": 0.135
                },
                {
                  "label": "AkashML (fp8)",
                  "value": 0.2
                },
                {
                  "label": "Parasail (fp8)",
                  "value": 0.22
                },
                {
                  "label": "SambaNova",
                  "value": 0.45
                },
                {
                  "label": "Groq",
                  "value": 0.59
                },
                {
                  "label": "CoreWeave (fp16)",
                  "value": 0.71
                },
                {
                  "label": "Google Vertex",
                  "value": 0.72
                },
                {
                  "label": "Cloudflare (fp8)",
                  "value": 0.293
                },
                {
                  "label": "Together",
                  "value": 1.04
                }
              ]
            },
            {
              "name": "Output",
              "points": [
                {
                  "label": "DeepInfra (fp8)",
                  "value": 0.32
                },
                {
                  "label": "Novita (bf16)",
                  "value": 0.4
                },
                {
                  "label": "AkashML (fp8)",
                  "value": 0.52
                },
                {
                  "label": "Parasail (fp8)",
                  "value": 0.5
                },
                {
                  "label": "SambaNova",
                  "value": 0.9
                },
                {
                  "label": "Groq",
                  "value": 0.79
                },
                {
                  "label": "CoreWeave (fp16)",
                  "value": 0.71
                },
                {
                  "label": "Google Vertex",
                  "value": 0.72
                },
                {
                  "label": "Cloudflare (fp8)",
                  "value": 2.253
                },
                {
                  "label": "Together",
                  "value": 1.04
                }
              ]
            },
            {
              "name": "Cache read",
              "points": [
                {
                  "label": "AkashML (fp8)",
                  "value": 0.1
                },
                {
                  "label": "Parasail (fp8)",
                  "value": 0.11
                },
                {
                  "label": "Groq",
                  "value": 0.295
                },
                {
                  "label": "CoreWeave (fp16)",
                  "value": 0.71
                }
              ]
            }
          ],
          "note": "Prices reported by OpenRouter’s public API, snapshot 2026-10-06; third-party-reported, not measured by Agent. Sorted from the lowest to the highest blended price (3 input : 1 output). A parenthesis names the quantization the provider reported; lower precision or a shorter context can explain a lower price, so check the endpoint table. Flex, priority, fast and regional endpoints are left out here because they are priced differently on purpose.",
          "factContext": "Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "sourceIds": [
            "openrouter-api-snapshot"
          ]
        },
        {
          "id": "provider-prices-deepseek-v4-pro",
          "title": "DeepSeek V4 Pro 0423: price per million tokens by provider",
          "subtitle": "Standard tier, one bar per provider (its cheapest standard endpoint); reported by OpenRouter’s public API, snapshot 2026-10-06",
          "kind": "grouped-bar",
          "unit": "usd",
          "yLabel": "USD per million tokens",
          "series": [
            {
              "name": "Input",
              "points": [
                {
                  "label": "StreamLake (fp8)",
                  "value": 0.2088
                },
                {
                  "label": "GMICloud (fp8)",
                  "value": 0.957
                },
                {
                  "label": "Relace (fp4)",
                  "value": 0.2067
                },
                {
                  "label": "Parasail (fp8)",
                  "value": 0.45
                },
                {
                  "label": "DigitalOcean",
                  "value": 1.044
                },
                {
                  "label": "Cloudflare",
                  "value": 1.15
                },
                {
                  "label": "DeepInfra (fp8)",
                  "value": 1.3
                },
                {
                  "label": "Alibaba (fp8)",
                  "value": 1.416
                },
                {
                  "label": "SiliconFlow (fp8)",
                  "value": 1.50162
                },
                {
                  "label": "Novita (fp8)",
                  "value": 1.6
                },
                {
                  "label": "Venice",
                  "value": 1.65
                },
                {
                  "label": "AtlasCloud (fp4)",
                  "value": 1.68
                },
                {
                  "label": "Baidu (fp8)",
                  "value": 1.69
                },
                {
                  "label": "NextBit (fp8)",
                  "value": 1.74
                },
                {
                  "label": "Reka",
                  "value": 0.9
                }
              ]
            },
            {
              "name": "Output",
              "points": [
                {
                  "label": "StreamLake (fp8)",
                  "value": 0.4176
                },
                {
                  "label": "GMICloud (fp8)",
                  "value": 1.914
                },
                {
                  "label": "Relace (fp4)",
                  "value": 4.2
                },
                {
                  "label": "Parasail (fp8)",
                  "value": 3.48
                },
                {
                  "label": "DigitalOcean",
                  "value": 2.088
                },
                {
                  "label": "Cloudflare",
                  "value": 2.55
                },
                {
                  "label": "DeepInfra (fp8)",
                  "value": 2.6
                },
                {
                  "label": "Alibaba (fp8)",
                  "value": 2.832
                },
                {
                  "label": "SiliconFlow (fp8)",
                  "value": 3.135
                },
                {
                  "label": "Novita (fp8)",
                  "value": 3.2
                },
                {
                  "label": "Venice",
                  "value": 3.301
                },
                {
                  "label": "AtlasCloud (fp4)",
                  "value": 3.38
                },
                {
                  "label": "Baidu (fp8)",
                  "value": 3.38
                },
                {
                  "label": "NextBit (fp8)",
                  "value": 3.48
                },
                {
                  "label": "Reka",
                  "value": 9
                }
              ]
            },
            {
              "name": "Cache read",
              "points": [
                {
                  "label": "StreamLake (fp8)",
                  "value": 0.0174
                },
                {
                  "label": "GMICloud (fp8)",
                  "value": 0.07975
                },
                {
                  "label": "Relace (fp4)",
                  "value": 0.21
                },
                {
                  "label": "Parasail (fp8)",
                  "value": 0.1
                },
                {
                  "label": "DigitalOcean",
                  "value": 0.2088
                },
                {
                  "label": "Cloudflare",
                  "value": 0.2
                },
                {
                  "label": "DeepInfra (fp8)",
                  "value": 0.1
                },
                {
                  "label": "Alibaba (fp8)",
                  "value": 0.118
                },
                {
                  "label": "SiliconFlow (fp8)",
                  "value": 0.135
                },
                {
                  "label": "Novita (fp8)",
                  "value": 0.135
                },
                {
                  "label": "Venice",
                  "value": 0.33
                },
                {
                  "label": "AtlasCloud (fp4)",
                  "value": 0.13
                },
                {
                  "label": "Baidu (fp8)",
                  "value": 0.14
                },
                {
                  "label": "NextBit (fp8)",
                  "value": 0.145
                },
                {
                  "label": "Reka",
                  "value": 0.18
                }
              ]
            }
          ],
          "note": "Prices reported by OpenRouter’s public API, snapshot 2026-10-06; third-party-reported, not measured by Agent. Sorted from the lowest to the highest blended price (3 input : 1 output). A parenthesis names the quantization the provider reported; lower precision or a shorter context can explain a lower price, so check the endpoint table. Flex, priority, fast and regional endpoints are left out here because they are priced differently on purpose.",
          "factContext": "DeepSeek V4 Pro 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "sourceIds": [
            "openrouter-api-snapshot"
          ]
        },
        {
          "id": "provider-prices-deepseek-v4-flash",
          "title": "DeepSeek V4 Flash 0423: price per million tokens by provider",
          "subtitle": "Standard tier, one bar per provider (its cheapest standard endpoint); reported by OpenRouter’s public API, snapshot 2026-10-06",
          "kind": "grouped-bar",
          "unit": "usd",
          "yLabel": "USD per million tokens",
          "series": [
            {
              "name": "Input",
              "points": [
                {
                  "label": "StreamLake (fp8)",
                  "value": 0.042
                },
                {
                  "label": "DeepInfra (fp8)",
                  "value": 0.09
                },
                {
                  "label": "GMICloud (fp8)",
                  "value": 0.091
                },
                {
                  "label": "Venice",
                  "value": 0.0966
                },
                {
                  "label": "DigitalOcean",
                  "value": 0.098
                },
                {
                  "label": "Alibaba (fp8)",
                  "value": 0.134
                },
                {
                  "label": "SiliconFlow (fp8)",
                  "value": 0.13
                },
                {
                  "label": "AtlasCloud (fp4)",
                  "value": 0.14
                },
                {
                  "label": "Baidu (fp8)",
                  "value": 0.14
                },
                {
                  "label": "Novita (fp8)",
                  "value": 0.14
                },
                {
                  "label": "Parasail (fp8)",
                  "value": 0.14
                },
                {
                  "label": "Mancer 2 (fp8)",
                  "value": 0.19
                },
                {
                  "label": "Relace (fp4)",
                  "value": 0.012
                },
                {
                  "label": "OpenInference (fp4)",
                  "value": 0.0132
                },
                {
                  "label": "Cloudflare",
                  "value": 0.44
                }
              ]
            },
            {
              "name": "Output",
              "points": [
                {
                  "label": "StreamLake (fp8)",
                  "value": 0.084
                },
                {
                  "label": "DeepInfra (fp8)",
                  "value": 0.18
                },
                {
                  "label": "GMICloud (fp8)",
                  "value": 0.182
                },
                {
                  "label": "Venice",
                  "value": 0.1925
                },
                {
                  "label": "DigitalOcean",
                  "value": 0.196
                },
                {
                  "label": "Alibaba (fp8)",
                  "value": 0.268
                },
                {
                  "label": "SiliconFlow (fp8)",
                  "value": 0.28
                },
                {
                  "label": "AtlasCloud (fp4)",
                  "value": 0.28
                },
                {
                  "label": "Baidu (fp8)",
                  "value": 0.28
                },
                {
                  "label": "Novita (fp8)",
                  "value": 0.28
                },
                {
                  "label": "Parasail (fp8)",
                  "value": 0.28
                },
                {
                  "label": "Mancer 2 (fp8)",
                  "value": 0.5
                },
                {
                  "label": "Relace (fp4)",
                  "value": 1.28
                },
                {
                  "label": "OpenInference (fp4)",
                  "value": 1.408
                },
                {
                  "label": "Cloudflare",
                  "value": 1.32
                }
              ]
            },
            {
              "name": "Cache read",
              "points": [
                {
                  "label": "StreamLake (fp8)",
                  "value": 0.0084
                },
                {
                  "label": "DeepInfra (fp8)",
                  "value": 0.018
                },
                {
                  "label": "GMICloud (fp8)",
                  "value": 0.0182
                },
                {
                  "label": "Venice",
                  "value": 0.0196
                },
                {
                  "label": "DigitalOcean",
                  "value": 0.0196
                },
                {
                  "label": "Alibaba (fp8)",
                  "value": 0.0268
                },
                {
                  "label": "SiliconFlow (fp8)",
                  "value": 0.028
                },
                {
                  "label": "AtlasCloud (fp4)",
                  "value": 0.028
                },
                {
                  "label": "Baidu (fp8)",
                  "value": 0.028
                },
                {
                  "label": "Novita (fp8)",
                  "value": 0.028
                },
                {
                  "label": "Parasail (fp8)",
                  "value": 0.07
                },
                {
                  "label": "Relace (fp4)",
                  "value": 0.012
                },
                {
                  "label": "OpenInference (fp4)",
                  "value": 0.0132
                },
                {
                  "label": "Cloudflare",
                  "value": 0.014
                }
              ]
            }
          ],
          "note": "Prices reported by OpenRouter’s public API, snapshot 2026-10-06; third-party-reported, not measured by Agent. Sorted from the lowest to the highest blended price (3 input : 1 output). A parenthesis names the quantization the provider reported; lower precision or a shorter context can explain a lower price, so check the endpoint table. Flex, priority, fast and regional endpoints are left out here because they are priced differently on purpose.",
          "factContext": "DeepSeek V4 Flash 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "sourceIds": [
            "openrouter-api-snapshot"
          ]
        },
        {
          "id": "provider-prices-kimi-k3",
          "title": "Kimi K3: price per million tokens by provider",
          "subtitle": "Standard tier, one bar per provider (its cheapest standard endpoint); reported by OpenRouter’s public API, snapshot 2026-10-06",
          "kind": "grouped-bar",
          "unit": "usd",
          "yLabel": "USD per million tokens",
          "series": [
            {
              "name": "Input",
              "points": [
                {
                  "label": "Relace (fp4)",
                  "value": 0.83
                },
                {
                  "label": "Phala",
                  "value": 1.95
                },
                {
                  "label": "Sail Research (fp4)",
                  "value": 0.84
                },
                {
                  "label": "Decart (mxfp4)",
                  "value": 2.01
                },
                {
                  "label": "InferenceNet (fp4)",
                  "value": 0.95
                },
                {
                  "label": "Wafer",
                  "value": 0.95
                },
                {
                  "label": "Morph (fp8)",
                  "value": 1.274
                },
                {
                  "label": "Makora",
                  "value": 1.53
                },
                {
                  "label": "AkashML (fp4)",
                  "value": 1.3
                },
                {
                  "label": "DigitalOcean",
                  "value": 2.55
                },
                {
                  "label": "Together",
                  "value": 2.7
                },
                {
                  "label": "DeepInfra (mxfp4)",
                  "value": 2.85
                },
                {
                  "label": "BaseTen (fp8)",
                  "value": 3
                },
                {
                  "label": "Chutes (mxfp4)",
                  "value": 3
                },
                {
                  "label": "Fireworks",
                  "value": 3
                },
                {
                  "label": "Modal (mxfp4)",
                  "value": 3
                },
                {
                  "label": "Moonshot AI (mxfp4)",
                  "value": 3
                },
                {
                  "label": "Parasail (fp4)",
                  "value": 3
                },
                {
                  "label": "Alibaba",
                  "value": 3.45
                }
              ]
            },
            {
              "name": "Output",
              "points": [
                {
                  "label": "Relace (fp4)",
                  "value": 13
                },
                {
                  "label": "Phala",
                  "value": 9.75
                },
                {
                  "label": "Sail Research (fp4)",
                  "value": 13.5
                },
                {
                  "label": "Decart (mxfp4)",
                  "value": 10.05
                },
                {
                  "label": "InferenceNet (fp4)",
                  "value": 14
                },
                {
                  "label": "Wafer",
                  "value": 14
                },
                {
                  "label": "Morph (fp8)",
                  "value": 13.296
                },
                {
                  "label": "Makora",
                  "value": 12.75
                },
                {
                  "label": "AkashML (fp4)",
                  "value": 14
                },
                {
                  "label": "DigitalOcean",
                  "value": 12.95
                },
                {
                  "label": "Together",
                  "value": 13.5
                },
                {
                  "label": "DeepInfra (mxfp4)",
                  "value": 14.25
                },
                {
                  "label": "BaseTen (fp8)",
                  "value": 15
                },
                {
                  "label": "Chutes (mxfp4)",
                  "value": 15
                },
                {
                  "label": "Fireworks",
                  "value": 15
                },
                {
                  "label": "Modal (mxfp4)",
                  "value": 15
                },
                {
                  "label": "Moonshot AI (mxfp4)",
                  "value": 15
                },
                {
                  "label": "Parasail (fp4)",
                  "value": 15
                },
                {
                  "label": "Alibaba",
                  "value": 17.25
                }
              ]
            },
            {
              "name": "Cache read",
              "points": [
                {
                  "label": "Relace (fp4)",
                  "value": 0.45
                },
                {
                  "label": "Phala",
                  "value": 0.195
                },
                {
                  "label": "Sail Research (fp4)",
                  "value": 0.3
                },
                {
                  "label": "Decart (mxfp4)",
                  "value": 0.201
                },
                {
                  "label": "InferenceNet (fp4)",
                  "value": 0.31
                },
                {
                  "label": "Wafer",
                  "value": 0.4
                },
                {
                  "label": "Morph (fp8)",
                  "value": 0.278
                },
                {
                  "label": "Makora",
                  "value": 0.204
                },
                {
                  "label": "AkashML (fp4)",
                  "value": 1.3
                },
                {
                  "label": "DigitalOcean",
                  "value": 0.255
                },
                {
                  "label": "Together",
                  "value": 0.27
                },
                {
                  "label": "DeepInfra (mxfp4)",
                  "value": 0.285
                },
                {
                  "label": "BaseTen (fp8)",
                  "value": 0.3
                },
                {
                  "label": "Chutes (mxfp4)",
                  "value": 0.3
                },
                {
                  "label": "Fireworks",
                  "value": 0.3
                },
                {
                  "label": "Modal (mxfp4)",
                  "value": 0.3
                },
                {
                  "label": "Moonshot AI (mxfp4)",
                  "value": 0.3
                },
                {
                  "label": "Parasail (fp4)",
                  "value": 0.3
                },
                {
                  "label": "Alibaba",
                  "value": 0.345
                }
              ]
            }
          ],
          "note": "Prices reported by OpenRouter’s public API, snapshot 2026-10-06; third-party-reported, not measured by Agent. Sorted from the lowest to the highest blended price (3 input : 1 output). A parenthesis names the quantization the provider reported; lower precision or a shorter context can explain a lower price, so check the endpoint table. Flex, priority, fast and regional endpoints are left out here because they are priced differently on purpose.",
          "factContext": "Kimi K3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "sourceIds": [
            "openrouter-api-snapshot"
          ]
        },
        {
          "id": "provider-prices-glm-5-3",
          "title": "GLM 5.3: price per million tokens by provider",
          "subtitle": "Standard tier, one bar per provider (its cheapest standard endpoint); reported by OpenRouter’s public API, snapshot 2026-10-06",
          "kind": "grouped-bar",
          "unit": "usd",
          "yLabel": "USD per million tokens",
          "series": [
            {
              "name": "Input",
              "points": [
                {
                  "label": "Novita (fp8)",
                  "value": 0.42
                },
                {
                  "label": "Reka",
                  "value": 0.17
                },
                {
                  "label": "Sail Research (fp8)",
                  "value": 0.2
                },
                {
                  "label": "Morph (fp8)",
                  "value": 0.179
                },
                {
                  "label": "DeepInfra (fp4)",
                  "value": 0.5625
                },
                {
                  "label": "SiliconFlow (fp8)",
                  "value": 0.7
                },
                {
                  "label": "InferenceNet",
                  "value": 0.14
                },
                {
                  "label": "Makora (fp4)",
                  "value": 0.18
                },
                {
                  "label": "AkashML (fp8)",
                  "value": 0.19
                },
                {
                  "label": "Phala",
                  "value": 0.84
                },
                {
                  "label": "Inceptron (fp4)",
                  "value": 0.6
                },
                {
                  "label": "DigitalOcean",
                  "value": 0.91
                },
                {
                  "label": "GMICloud (fp8)",
                  "value": 0.98
                },
                {
                  "label": "Alibaba",
                  "value": 1.19
                },
                {
                  "label": "Decart (fp4)",
                  "value": 1.19
                },
                {
                  "label": "Wafer",
                  "value": 0.15
                },
                {
                  "label": "Friendli",
                  "value": 1.26
                },
                {
                  "label": "AtlasCloud (fp8)",
                  "value": 1.4
                },
                {
                  "label": "Baidu (fp8)",
                  "value": 1.4
                },
                {
                  "label": "BaseTen (fp4)",
                  "value": 1.4
                },
                {
                  "label": "Cloudflare",
                  "value": 1.4
                },
                {
                  "label": "Crusoe (fp4)",
                  "value": 1.4
                },
                {
                  "label": "Fireworks",
                  "value": 1.4
                },
                {
                  "label": "Mistral (nvfp4)",
                  "value": 1.4
                },
                {
                  "label": "Modal",
                  "value": 1.4
                },
                {
                  "label": "Nebius (fp4)",
                  "value": 1.4
                },
                {
                  "label": "Parasail (fp8)",
                  "value": 1.4
                },
                {
                  "label": "PrimeIntellect",
                  "value": 1.4
                },
                {
                  "label": "Together",
                  "value": 1.4
                },
                {
                  "label": "Venice",
                  "value": 1.4
                },
                {
                  "label": "Z.AI (fp8)",
                  "value": 1.4
                },
                {
                  "label": "Relace",
                  "value": 0.03
                }
              ]
            },
            {
              "name": "Output",
              "points": [
                {
                  "label": "Novita (fp8)",
                  "value": 1.32
                },
                {
                  "label": "Reka",
                  "value": 3
                },
                {
                  "label": "Sail Research (fp8)",
                  "value": 3.4
                },
                {
                  "label": "Morph (fp8)",
                  "value": 3.553
                },
                {
                  "label": "DeepInfra (fp4)",
                  "value": 2.5
                },
                {
                  "label": "SiliconFlow (fp8)",
                  "value": 2.2
                },
                {
                  "label": "InferenceNet",
                  "value": 4.4
                },
                {
                  "label": "Makora (fp4)",
                  "value": 4.4
                },
                {
                  "label": "AkashML (fp8)",
                  "value": 4.4
                },
                {
                  "label": "Phala",
                  "value": 2.64
                },
                {
                  "label": "Inceptron (fp4)",
                  "value": 3.39
                },
                {
                  "label": "DigitalOcean",
                  "value": 2.86
                },
                {
                  "label": "GMICloud (fp8)",
                  "value": 3.08
                },
                {
                  "label": "Alibaba",
                  "value": 3.74
                },
                {
                  "label": "Decart (fp4)",
                  "value": 3.74
                },
                {
                  "label": "Wafer",
                  "value": 7
                },
                {
                  "label": "Friendli",
                  "value": 3.96
                },
                {
                  "label": "AtlasCloud (fp8)",
                  "value": 4.4
                },
                {
                  "label": "Baidu (fp8)",
                  "value": 4.4
                },
                {
                  "label": "BaseTen (fp4)",
                  "value": 4.4
                },
                {
                  "label": "Cloudflare",
                  "value": 4.4
                },
                {
                  "label": "Crusoe (fp4)",
                  "value": 4.4
                },
                {
                  "label": "Fireworks",
                  "value": 4.4
                },
                {
                  "label": "Mistral (nvfp4)",
                  "value": 4.4
                },
                {
                  "label": "Modal",
                  "value": 4.4
                },
                {
                  "label": "Nebius (fp4)",
                  "value": 4.4
                },
                {
                  "label": "Parasail (fp8)",
                  "value": 4.4
                },
                {
                  "label": "PrimeIntellect",
                  "value": 4.4
                },
                {
                  "label": "Together",
                  "value": 4.4
                },
                {
                  "label": "Venice",
                  "value": 4.4
                },
                {
                  "label": "Z.AI (fp8)",
                  "value": 4.4
                },
                {
                  "label": "Relace",
                  "value": 12
                }
              ]
            },
            {
              "name": "Cache read",
              "points": [
                {
                  "label": "Novita (fp8)",
                  "value": 0.078
                },
                {
                  "label": "Reka",
                  "value": 0.169
                },
                {
                  "label": "Sail Research (fp8)",
                  "value": 0.15
                },
                {
                  "label": "Morph (fp8)",
                  "value": 0.137
                },
                {
                  "label": "DeepInfra (fp4)",
                  "value": 0.125
                },
                {
                  "label": "SiliconFlow (fp8)",
                  "value": 0.13
                },
                {
                  "label": "InferenceNet",
                  "value": 0.07
                },
                {
                  "label": "Makora (fp4)",
                  "value": 0.19
                },
                {
                  "label": "AkashML (fp8)",
                  "value": 0.19
                },
                {
                  "label": "Phala",
                  "value": 0.156
                },
                {
                  "label": "Inceptron (fp4)",
                  "value": 0.2
                },
                {
                  "label": "DigitalOcean",
                  "value": 0.169
                },
                {
                  "label": "GMICloud (fp8)",
                  "value": 0.182
                },
                {
                  "label": "Alibaba",
                  "value": 0.238
                },
                {
                  "label": "Decart (fp4)",
                  "value": 0.1955
                },
                {
                  "label": "Wafer",
                  "value": 0.14
                },
                {
                  "label": "Friendli",
                  "value": 0.234
                },
                {
                  "label": "AtlasCloud (fp8)",
                  "value": 0.26
                },
                {
                  "label": "Baidu (fp8)",
                  "value": 0.26
                },
                {
                  "label": "BaseTen (fp4)",
                  "value": 0.14
                },
                {
                  "label": "Cloudflare",
                  "value": 0.26
                },
                {
                  "label": "Crusoe (fp4)",
                  "value": 0.26
                },
                {
                  "label": "Fireworks",
                  "value": 0.26
                },
                {
                  "label": "Mistral (nvfp4)",
                  "value": 0.14
                },
                {
                  "label": "Modal",
                  "value": 0.26
                },
                {
                  "label": "Parasail (fp8)",
                  "value": 0.26
                },
                {
                  "label": "PrimeIntellect",
                  "value": 0.26
                },
                {
                  "label": "Together",
                  "value": 0.26
                },
                {
                  "label": "Venice",
                  "value": 0.26
                },
                {
                  "label": "Z.AI (fp8)",
                  "value": 0.26
                },
                {
                  "label": "Relace",
                  "value": 0.03
                }
              ]
            }
          ],
          "note": "Prices reported by OpenRouter’s public API, snapshot 2026-10-06; third-party-reported, not measured by Agent. Sorted from the lowest to the highest blended price (3 input : 1 output). A parenthesis names the quantization the provider reported; lower precision or a shorter context can explain a lower price, so check the endpoint table. Flex, priority, fast and regional endpoints are left out here because they are priced differently on purpose.",
          "factContext": "GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "sourceIds": [
            "openrouter-api-snapshot"
          ]
        }
      ],
      "tables": [
        {
          "id": "provider-index-models",
          "title": "Provider index per model (reported by OpenRouter’s public API, snapshot 2026-10-06)",
          "columns": [
            {
              "key": "model",
              "label": "Model",
              "unit": "text"
            },
            {
              "key": "providers",
              "label": "Providers",
              "unit": "count"
            },
            {
              "key": "endpoints",
              "label": "Endpoints (all tiers)",
              "unit": "count"
            },
            {
              "key": "listPrice",
              "label": "OpenRouter list price, in / out (USD per M)",
              "unit": "text"
            },
            {
              "key": "cheapest",
              "label": "Cheapest standard provider",
              "unit": "text"
            },
            {
              "key": "priciest",
              "label": "Most expensive standard provider",
              "unit": "text"
            },
            {
              "key": "spread",
              "label": "Spread (blended)",
              "unit": "text"
            },
            {
              "key": "firstParty",
              "label": "First-party list price, in / out",
              "unit": "text"
            }
          ],
          "rows": [
            {
              "model": "Claude Haiku 4.5",
              "providers": 4,
              "endpoints": 8,
              "listPrice": "$1.00 / $5.00",
              "cheapest": "Amazon Bedrock: $1.00 / $5.00",
              "priciest": "Google Vertex: $1.00 / $5.00",
              "spread": "1.0x",
              "firstParty": "Anthropic: $1.00 / $5.00"
            },
            {
              "model": "Claude Sonnet 5",
              "providers": 5,
              "endpoints": 10,
              "listPrice": "$2.00 / $10.00",
              "cheapest": "Amazon Bedrock: $2.00 / $10.00",
              "priciest": "Google Vertex: $2.00 / $10.00",
              "spread": "1.0x",
              "firstParty": "Anthropic: $2.00 / $10.00"
            },
            {
              "model": "Claude Sonnet 5.5",
              "providers": 5,
              "endpoints": 8,
              "listPrice": "$2.00 / $10.00",
              "cheapest": "Amazon Bedrock: $2.00 / $10.00",
              "priciest": "Google Vertex: $2.00 / $10.00",
              "spread": "1.0x",
              "firstParty": "Anthropic: $2.00 / $10.00"
            },
            {
              "model": "Claude Opus 4.8",
              "providers": 5,
              "endpoints": 11,
              "listPrice": "$5.00 / $25.00",
              "cheapest": "Amazon Bedrock: $5.00 / $25.00",
              "priciest": "Google Vertex: $5.00 / $25.00",
              "spread": "1.0x",
              "firstParty": "Anthropic: $5.00 / $25.00"
            },
            {
              "model": "Claude Opus 5",
              "providers": 5,
              "endpoints": 11,
              "listPrice": "$5.00 / $25.00",
              "cheapest": "Amazon Bedrock: $5.00 / $25.00",
              "priciest": "Google Vertex: $5.00 / $25.00",
              "spread": "1.0x",
              "firstParty": "Anthropic: $5.00 / $25.00"
            },
            {
              "model": "Claude Opus 5.5",
              "providers": 5,
              "endpoints": 11,
              "listPrice": "$4.00 / $20.00",
              "cheapest": "Amazon Bedrock: $4.00 / $20.00",
              "priciest": "Google Vertex: $4.00 / $20.00",
              "spread": "1.0x",
              "firstParty": "Anthropic: $4.00 / $20.00"
            },
            {
              "model": "Claude Fable 5.1",
              "providers": 4,
              "endpoints": 4,
              "listPrice": "$10.00 / $50.00",
              "cheapest": "Amazon Bedrock: $10.00 / $50.00",
              "priciest": "Google Vertex: $10.00 / $50.00",
              "spread": "1.0x",
              "firstParty": "Anthropic: $10.00 / $50.00"
            },
            {
              "model": "GPT-6 Sol",
              "providers": 3,
              "endpoints": 7,
              "listPrice": "$2.00 / $10.00",
              "cheapest": "Azure: $2.00 / $10.00",
              "priciest": "OpenAI: $2.00 / $10.00",
              "spread": "1.0x",
              "firstParty": "not in the price table"
            },
            {
              "model": "GPT-6 Luna",
              "providers": 3,
              "endpoints": 7,
              "listPrice": "$0.10 / $0.50",
              "cheapest": "Azure: $0.10 / $0.50",
              "priciest": "OpenAI: $0.10 / $0.50",
              "spread": "1.0x",
              "firstParty": "OpenAI: $0.10 / $0.50"
            },
            {
              "model": "GPT-6 Astra",
              "providers": 3,
              "endpoints": 7,
              "listPrice": "$10.00 / $50.00",
              "cheapest": "Azure: $10.00 / $50.00",
              "priciest": "OpenAI: $10.00 / $50.00",
              "spread": "1.0x",
              "firstParty": "not in the price table"
            },
            {
              "model": "GPT-5.5",
              "providers": 3,
              "endpoints": 7,
              "listPrice": "$5.00 / $30.00",
              "cheapest": "Azure: $5.00 / $30.00",
              "priciest": "OpenAI: $5.00 / $30.00",
              "spread": "1.0x",
              "firstParty": "not in the price table"
            },
            {
              "model": "gpt-oss-120b",
              "providers": 20,
              "endpoints": 23,
              "listPrice": "$0.037 / $0.17",
              "cheapest": "CoreWeave (fp4): $0.03 / $0.17",
              "priciest": "Cerebras (fp16): $0.35 / $0.75",
              "spread": "6.9x",
              "firstParty": "not in the price table"
            },
            {
              "model": "Gemini 3.8 Flash",
              "providers": 2,
              "endpoints": 6,
              "listPrice": "$0.75 / $3.75",
              "cheapest": "Google AI Studio: $0.75 / $3.75",
              "priciest": "Google Vertex: $0.75 / $3.75",
              "spread": "1.0x",
              "firstParty": "Google: $0.75 / $3.75"
            },
            {
              "model": "Gemini 3.5 Flash",
              "providers": 2,
              "endpoints": 7,
              "listPrice": "$1.50 / $9.00",
              "cheapest": "Google AI Studio: $1.50 / $9.00",
              "priciest": "Google Vertex: $1.50 / $9.00",
              "spread": "1.0x",
              "firstParty": "Google: $1.50 / $9.00"
            },
            {
              "model": "Gemini 3.5 Flash Lite",
              "providers": 2,
              "endpoints": 8,
              "listPrice": "$0.30 / $2.50",
              "cheapest": "Google AI Studio: $0.30 / $2.50",
              "priciest": "Google Vertex: $0.30 / $2.50",
              "spread": "1.0x",
              "firstParty": "not in the price table"
            },
            {
              "model": "Gemini 3.1 Pro Preview",
              "providers": 2,
              "endpoints": 6,
              "listPrice": "$2.00 / $12.00",
              "cheapest": "Google AI Studio: $2.00 / $12.00",
              "priciest": "Google Vertex: $2.00 / $12.00",
              "spread": "1.0x",
              "firstParty": "not in the price table"
            },
            {
              "model": "Llama 4 Maverick",
              "providers": 4,
              "endpoints": 4,
              "listPrice": "$0.19 / $0.65",
              "cheapest": "DigitalOcean: $0.19 / $0.65",
              "priciest": "Parasail (fp8): $0.35 / $1.00",
              "spread": "1.7x",
              "firstParty": "not in the price table"
            },
            {
              "model": "Llama 3.3 70B Instruct",
              "providers": 10,
              "endpoints": 11,
              "listPrice": "$0.22 / $0.50",
              "cheapest": "DeepInfra (fp8): $0.10 / $0.32",
              "priciest": "Together: $1.04 / $1.04",
              "spread": "6.7x",
              "firstParty": "not in the price table"
            },
            {
              "model": "DeepSeek V4 Pro 0423",
              "providers": 16,
              "endpoints": 16,
              "listPrice": "$0.21 / $0.42",
              "cheapest": "StreamLake (fp8): $0.21 / $0.42",
              "priciest": "Reka: $0.90 / $9.00",
              "spread": "11.2x",
              "firstParty": "not in the price table"
            },
            {
              "model": "DeepSeek V4 Flash 0423",
              "providers": 16,
              "endpoints": 16,
              "listPrice": "$0.012 / $1.28",
              "cheapest": "StreamLake (fp8): $0.042 / $0.084",
              "priciest": "Cloudflare: $0.44 / $1.32",
              "spread": "12.6x",
              "firstParty": "not in the price table"
            },
            {
              "model": "Qwen3.8 Max (0902)",
              "providers": 1,
              "endpoints": 1,
              "listPrice": "$2.00 / $6.00",
              "cheapest": "Alibaba: $2.00 / $6.00",
              "priciest": "only one provider",
              "spread": "n/a",
              "firstParty": "not in the price table"
            },
            {
              "model": "Qwen3.8 Flash",
              "providers": 1,
              "endpoints": 1,
              "listPrice": "$0.15 / $0.47",
              "cheapest": "Alibaba: $0.15 / $0.47",
              "priciest": "only one provider",
              "spread": "n/a",
              "firstParty": "not in the price table"
            },
            {
              "model": "Mistral Medium 3.5",
              "providers": 1,
              "endpoints": 3,
              "listPrice": "$1.50 / $7.50",
              "cheapest": "Mistral: $1.50 / $7.50",
              "priciest": "only one provider",
              "spread": "n/a",
              "firstParty": "not in the price table"
            },
            {
              "model": "Mistral Large 3 2512",
              "providers": 1,
              "endpoints": 2,
              "listPrice": "$0.50 / $1.50",
              "cheapest": "Mistral: $0.50 / $1.50",
              "priciest": "only one provider",
              "spread": "n/a",
              "firstParty": "not in the price table"
            },
            {
              "model": "Grok 4.7",
              "providers": 1,
              "endpoints": 5,
              "listPrice": "$2.00 / $6.00",
              "cheapest": "xAI: $2.00 / $6.00",
              "priciest": "only one provider",
              "spread": "n/a",
              "firstParty": "not in the price table"
            },
            {
              "model": "Kimi K3",
              "providers": 20,
              "endpoints": 24,
              "listPrice": "$0.95 / $14.00",
              "cheapest": "Relace (fp4): $0.83 / $13.00",
              "priciest": "Alibaba: $3.45 / $17.25",
              "spread": "1.8x",
              "firstParty": "not in the price table"
            },
            {
              "model": "GLM 5.3",
              "providers": 32,
              "endpoints": 41,
              "listPrice": "$0.07 / $7.00",
              "cheapest": "Novita (fp8): $0.42 / $1.32",
              "priciest": "Relace: $0.03 / $12.00",
              "spread": "4.7x",
              "firstParty": "not in the price table"
            }
          ]
        },
        {
          "id": "gateway-overhead-status",
          "title": "Gateway overhead: what is and is not measured",
          "columns": [
            {
              "key": "metric",
              "label": "Metric",
              "unit": "text"
            },
            {
              "key": "status",
              "label": "Status",
              "unit": "text"
            }
          ],
          "rows": [
            {
              "metric": "Per-token price through OpenRouter vs first-party",
              "status": "Reported list prices, snapshot 2026-10-06"
            },
            {
              "metric": "Credit-purchase fee",
              "status": "5.5% on Standard (third-party-reported, 2026-10-06)"
            },
            {
              "metric": "Provider latency and throughput (last 30 min)",
              "status": "unknown: the keyless API returned none for 265 endpoints"
            },
            {
              "metric": "Time to first token and total time through the gateway vs direct",
              "status": "not measured: no key in the environment. The live harness sends nothing without an OpenRouter key."
            },
            {
              "metric": "Billed cost per call through the gateway",
              "status": "not measured: no key in the environment. The live harness sends nothing without an OpenRouter key."
            }
          ]
        },
        {
          "id": "provider-index-endpoints",
          "title": "Every endpoint OpenRouter listed (reported by OpenRouter’s public API, snapshot 2026-10-06)",
          "columns": [
            {
              "key": "model",
              "label": "Model",
              "unit": "text"
            },
            {
              "key": "provider",
              "label": "Provider",
              "unit": "text"
            },
            {
              "key": "tier",
              "label": "Tier",
              "unit": "text"
            },
            {
              "key": "quantization",
              "label": "Quantization",
              "unit": "text"
            },
            {
              "key": "context",
              "label": "Context (tokens)",
              "unit": "tokens"
            },
            {
              "key": "input",
              "label": "Input (USD per M)",
              "unit": "usd"
            },
            {
              "key": "output",
              "label": "Output (USD per M)",
              "unit": "usd"
            },
            {
              "key": "cacheRead",
              "label": "Cache read (USD per M)",
              "unit": "usd"
            },
            {
              "key": "uptime1d",
              "label": "Uptime, last day (%)",
              "unit": "percent"
            }
          ],
          "rows": [
            {
              "model": "Claude Fable 5.1",
              "provider": "Azure",
              "tier": "standard",
              "quantization": "not reported",
              "context": 1000000,
              "input": 10,
              "output": 50,
              "cacheRead": 0.25,
              "uptime1d": 99.36
            },
            {
              "model": "Claude Fable 5.1",
              "provider": "Anthropic",
              "tier": "standard",
              "quantization": "not reported",
              "context": 1000000,
              "input": 10,
              "output": 50,
              "cacheRead": 0.25,
              "uptime1d": 99.65
            },
            {
              "model": "Claude Fable 5.1",
              "provider": "Amazon Bedrock",
              "tier": "standard",
              "quantization": "not reported",
              "context": 1000000,
              "input": 10,
              "output": 50,
              "cacheRead": 0.25,
              "uptime1d": null
            },
            {
              "model": "Claude Fable 5.1",
              "provider": "Google Vertex",
              "tier": "standard",
              "quantization": "not reported",
              "context": 1000000,
              "input": 10,
              "output": 50,
              "cacheRead": 0.25,
              "uptime1d": 99.92
            },
            {
              "model": "Claude Haiku 4.5",
              "provider": "Azure",
              "tier": "standard",
              "quantization": "not reported",
              "context": 200000,
              "input": 1,
              "output": 5,
              "cacheRead": 0.1,
              "uptime1d": 99.93
            },
            {
              "model": "Claude Haiku 4.5",
              "provider": "Amazon Bedrock",
              "tier": "standard",
              "quantization": "not reported",
              "context": 200000,
              "input": 1,
              "output": 5,
              "cacheRead": 0.1,
              "uptime1d": 99.96
            },
            {
              "model": "Claude Haiku 4.5",
              "provider": "Google Vertex",
              "tier": "standard",
              "quantization": "not reported",
              "context": 200000,
              "input": 1,
              "output": 5,
              "cacheRead": 0.1,
              "uptime1d": 99.65
            },
            {
              "model": "Claude Haiku 4.5",
              "provider": "Anthropic",
              "tier": "standard",
              "quantization": "not reported",
              "context": 200000,
              "input": 1,
              "output": 5,
              "cacheRead": 0.1,
              "uptime1d": 99.97
            },
            {
              "model": "Claude Haiku 4.5",
              "provider": "Amazon Bedrock",
              "tier": "regional",
              "quantization": "not reported",
              "context": 200000,
              "input": 1.1,
              "output": 5.5,
              "cacheRead": 0.11,
              "uptime1d": 99.88
            },
            {
              "model": "Claude Haiku 4.5",
              "provider": "Google Vertex",
              "tier": "regional",
              "quantization": "not reported",
              "context": 200000,
              "input": 1.1,
              "output": 5.5,
              "cacheRead": 0.11,
              "uptime1d": 100
            },
            {
              "model": "Claude Haiku 4.5",
              "provider": "Amazon Bedrock",
              "tier": "regional",
              "quantization": "not reported",
              "context": 200000,
              "input": 1.1,
              "output": 5.5,
              "cacheRead": 0.11,
              "uptime1d": 99.74
            },
            {
              "model": "Claude Haiku 4.5",
              "provider": "Google Vertex",
              "tier": "regional",
              "quantization": "not reported",
              "context": 200000,
              "input": 1.1,
              "output": 5.5,
              "cacheRead": 0.11,
              "uptime1d": 100
            },
            {
              "model": "Claude Opus 4.8",
              "provider": "Claude Platform on AWS",
              "tier": "standard",
              "quantization": "not reported",
              "context": 1000000,
              "input": 5,
              "output": 25,
              "cacheRead": 0.5,
              "uptime1d": 99.99
            },
            {
              "model": "Claude Opus 4.8",
              "provider": "Azure",
              "tier": "standard",
              "quantization": "not reported",
              "context": 1000000,
              "input": 5,
              "output": 25,
              "cacheRead": 0.5,
              "uptime1d": 98.13
            },
            {
              "model": "Claude Opus 4.8",
              "provider": "Amazon Bedrock",
              "tier": "standard",
              "quantization": "not reported",
              "context": 1000000,
              "input": 5,
              "output": 25,
              "cacheRead": 0.5,
              "uptime1d": 94.89
            },
            {
              "model": "Claude Opus 4.8",
              "provider": "Google Vertex",
              "tier": "standard",
              "quantization": "not reported",
              "context": 1000000,
              "input": 5,
              "output": 25,
              "cacheRead": 0.5,
              "uptime1d": 99.97
            },
            {
              "model": "Claude Opus 4.8",
              "provider": "Anthropic",
              "tier": "standard",
              "quantization": "not reported",
              "context": 1000000,
              "input": 5,
              "output": 25,
              "cacheRead": 0.5,
              "uptime1d": 99.99
            },
            {
              "model": "Claude Opus 4.8",
              "provider": "Azure",
              "tier": "regional",
              "quantization": "not reported",
              "context": 1000000,
              "input": 5.5,
              "output": 27.5,
              "cacheRead": 0.55,
              "uptime1d": null
            },
            {
              "model": "Claude Opus 4.8",
              "provider": "Amazon Bedrock",
              "tier": "regional",
              "quantization": "not reported",
              "context": 1000000,
              "input": 5.5,
              "output": 27.5,
              "cacheRead": 0.55,
              "uptime1d": 100
            },
            {
              "model": "Claude Opus 4.8",
              "provider": "Google Vertex",
              "tier": "regional",
              "quantization": "not reported",
              "context": 1000000,
              "input": 5.5,
              "output": 27.5,
              "cacheRead": 0.55,
              "uptime1d": null
            },
            {
              "model": "Claude Opus 4.8",
              "provider": "Amazon Bedrock",
              "tier": "regional",
              "quantization": "not reported",
              "context": 1000000,
              "input": 5.5,
              "output": 27.5,
              "cacheRead": 0.55,
              "uptime1d": 100
            },
            {
              "model": "Claude Opus 4.8",
              "provider": "Google Vertex",
              "tier": "regional",
              "quantization": "not reported",
              "context": 1000000,
              "input": 5.5,
              "output": 27.5,
              "cacheRead": 0.55,
              "uptime1d": 100
            },
            {
              "model": "Claude Opus 4.8",
              "provider": "Anthropic",
              "tier": "fast",
              "quantization": "not reported",
              "context": 1000000,
              "input": 10,
              "output": 50,
              "cacheRead": 1,
              "uptime1d": 100
            },
            {
              "model": "Claude Opus 5",
              "provider": "Claude Platform on AWS",
              "tier": "standard",
              "quantization": "not reported",
              "context": 1000000,
              "input": 5,
              "output": 25,
              "cacheRead": 0.5,
              "uptime1d": 99.35
            },
            {
              "model": "Claude Opus 5",
              "provider": "Google Vertex",
              "tier": "standard",
              "quantization": "not reported",
              "context": 1000000,
              "input": 5,
              "output": 25,
              "cacheRead": 0.5,
              "uptime1d": 99.97
            },
            {
              "model": "Claude Opus 5",
              "provider": "Amazon Bedrock",
              "tier": "standard",
              "quantization": "not reported",
              "context": 1000000,
              "input": 5,
              "output": 25,
              "cacheRead": 0.5,
              "uptime1d": 99.94
            },
            {
              "model": "Claude Opus 5",
              "provider": "Azure",
              "tier": "standard",
              "quantization": "not reported",
              "context": 1000000,
              "input": 5,
              "output": 25,
              "cacheRead": 0.5,
              "uptime1d": 100
            },
            {
              "model": "Claude Opus 5",
              "provider": "Anthropic",
              "tier": "standard",
              "quantization": "not reported",
              "context": 1000000,
              "input": 5,
              "output": 25,
              "cacheRead": 0.5,
              "uptime1d": 99.86
            },
            {
              "model": "Claude Opus 5",
              "provider": "Amazon Bedrock",
              "tier": "regional",
              "quantization": "not reported",
              "context": 1000000,
              "input": 5.5,
              "output": 27.5,
              "cacheRead": 0.55,
              "uptime1d": 99.74
            },
            {
              "model": "Claude Opus 5",
              "provider": "Azure",
              "tier": "regional",
              "quantization": "not reported",
              "context": 1000000,
              "input": 5.5,
              "output": 27.5,
              "cacheRead": 0.55,
              "uptime1d": null
            },
            {
              "model": "Claude Opus 5",
              "provider": "Google Vertex",
              "tier": "regional",
              "quantization": "not reported",
              "context": 1000000,
              "input": 5.5,
              "output": 27.5,
              "cacheRead": 0.55,
              "uptime1d": 100
            },
            {
              "model": "Claude Opus 5",
              "provider": "Google Vertex",
              "tier": "regional",
              "quantization": "not reported",
              "context": 1000000,
              "input": 5.5,
              "output": 27.5,
              "cacheRead": 0.55,
              "uptime1d": 100
            },
            {
              "model": "Claude Opus 5",
              "provider": "Amazon Bedrock",
              "tier": "regional",
              "quantization": "not reported",
              "context": 1000000,
              "input": 5.5,
              "output": 27.5,
              "cacheRead": 0.55,
              "uptime1d": 100
            },
            {
              "model": "Claude Opus 5",
              "provider": "Anthropic",
              "tier": "fast",
              "quantization": "not reported",
              "context": 1000000,
              "input": 10,
              "output": 50,
              "cacheRead": 1,
              "uptime1d": 100
            },
            {
              "model": "Claude Opus 5.5",
              "provider": "Amazon Bedrock",
              "tier": "standard",
              "quantization": "not reported",
              "context": 1000000,
              "input": 4,
              "output": 20,
              "cacheRead": 0.2,
              "uptime1d": 97.21
            },
            {
              "model": "Claude Opus 5.5",
              "provider": "Azure",
              "tier": "standard",
              "quantization": "not reported",
              "context": 1000000,
              "input": 4,
              "output": 20,
              "cacheRead": 0.2,
              "uptime1d": 99.99
            },
            {
              "model": "Claude Opus 5.5",
              "provider": "Google Vertex",
              "tier": "standard",
              "quantization": "not reported",
              "context": 1000000,
              "input": 4,
              "output": 20,
              "cacheRead": 0.2,
              "uptime1d": 99.96
            },
            {
              "model": "Claude Opus 5.5",
              "provider": "Claude Platform on AWS",
              "tier": "standard",
              "quantization": "not reported",
              "context": 1000000,
              "input": 4,
              "output": 20,
              "cacheRead": 0.2,
              "uptime1d": 99.91
            },
            {
              "model": "Claude Opus 5.5",
              "provider": "Anthropic",
              "tier": "standard",
              "quantization": "not reported",
              "context": 1000000,
              "input": 4,
              "output": 20,
              "cacheRead": 0.2,
              "uptime1d": 99.94
            },
            {
              "model": "Claude Opus 5.5",
              "provider": "Amazon Bedrock",
              "tier": "regional",
              "quantization": "not reported",
              "context": 1000000,
              "input": 4.4,
              "output": 22,
              "cacheRead": 0.22,
              "uptime1d": 99.9
            },
            {
              "model": "Claude Opus 5.5",
              "provider": "Amazon Bedrock",
              "tier": "regional",
              "quantization": "not reported",
              "context": 1000000,
              "input": 4.4,
              "output": 22,
              "cacheRead": 0.22,
              "uptime1d": 99.84
            },
            {
              "model": "Claude Opus 5.5",
              "provider": "Azure",
              "tier": "regional",
              "quantization": "not reported",
              "context": 1000000,
              "input": 4.4,
              "output": 22,
              "cacheRead": 0.22,
              "uptime1d": 100
            },
            {
              "model": "Claude Opus 5.5",
              "provider": "Google Vertex",
              "tier": "regional",
              "quantization": "not reported",
              "context": 1000000,
              "input": 4.4,
              "output": 22,
              "cacheRead": 0.22,
              "uptime1d": 100
            },
            {
              "model": "Claude Opus 5.5",
              "provider": "Google Vertex",
              "tier": "regional",
              "quantization": "not reported",
              "context": 1000000,
              "input": 4.4,
              "output": 22,
              "cacheRead": 0.22,
              "uptime1d": 99.99
            },
            {
              "model": "Claude Opus 5.5",
              "provider": "Anthropic",
              "tier": "fast",
              "quantization": "not reported",
              "context": 1000000,
              "input": 8,
              "output": 40,
              "cacheRead": 0.4,
              "uptime1d": 100
            },
            {
              "model": "Claude Sonnet 5",
              "provider": "Claude Platform on AWS",
              "tier": "standard",
              "quantization": "not reported",
              "context": 1000000,
              "input": 2,
              "output": 10,
              "cacheRead": 0.2,
              "uptime1d": 100
            },
            {
              "model": "Claude Sonnet 5",
              "provider": "Azure",
              "tier": "standard",
              "quantization": "not reported",
              "context": 1000000,
              "input": 2,
              "output": 10,
              "cacheRead": 0.2,
              "uptime1d": 99.74
            },
            {
              "model": "Claude Sonnet 5",
              "provider": "Google Vertex",
              "tier": "standard",
              "quantization": "not reported",
              "context": 1000000,
              "input": 2,
              "output": 10,
              "cacheRead": 0.2,
              "uptime1d": 99.99
            },
            {
              "model": "Claude Sonnet 5",
              "provider": "Amazon Bedrock",
              "tier": "standard",
              "quantization": "not reported",
              "context": 1000000,
              "input": 2,
              "output": 10,
              "cacheRead": 0.2,
              "uptime1d": 99.99
            },
            {
              "model": "Claude Sonnet 5",
              "provider": "Anthropic",
              "tier": "standard",
              "quantization": "not reported",
              "context": 1000000,
              "input": 2,
              "output": 10,
              "cacheRead": 0.2,
              "uptime1d": 99.98
            },
            {
              "model": "Claude Sonnet 5",
              "provider": "Amazon Bedrock",
              "tier": "regional",
              "quantization": "not reported",
              "context": 1000000,
              "input": 2.2,
              "output": 11,
              "cacheRead": 0.22,
              "uptime1d": 100
            },
            {
              "model": "Claude Sonnet 5",
              "provider": "Azure",
              "tier": "regional",
              "quantization": "not reported",
              "context": 1000000,
              "input": 2.2,
              "output": 11,
              "cacheRead": 0.22,
              "uptime1d": null
            },
            {
              "model": "Claude Sonnet 5",
              "provider": "Google Vertex",
              "tier": "regional",
              "quantization": "not reported",
              "context": 1000000,
              "input": 2.2,
              "output": 11,
              "cacheRead": 0.22,
              "uptime1d": 100
            },
            {
              "model": "Claude Sonnet 5",
              "provider": "Google Vertex",
              "tier": "regional",
              "quantization": "not reported",
              "context": 1000000,
              "input": 2.2,
              "output": 11,
              "cacheRead": 0.22,
              "uptime1d": 99.94
            },
            {
              "model": "Claude Sonnet 5",
              "provider": "Amazon Bedrock",
              "tier": "regional",
              "quantization": "not reported",
              "context": 1000000,
              "input": 2.2,
              "output": 11,
              "cacheRead": 0.22,
              "uptime1d": 100
            },
            {
              "model": "Claude Sonnet 5.5",
              "provider": "Google Vertex",
              "tier": "standard",
              "quantization": "not reported",
              "context": 1000000,
              "input": 2,
              "output": 10,
              "cacheRead": 0.2,
              "uptime1d": 99.99
            },
            {
              "model": "Claude Sonnet 5.5",
              "provider": "Amazon Bedrock",
              "tier": "standard",
              "quantization": "not reported",
              "context": 1000000,
              "input": 2,
              "output": 10,
              "cacheRead": 0.2,
              "uptime1d": 99.98
            },
            {
              "model": "Claude Sonnet 5.5",
              "provider": "Azure",
              "tier": "standard",
              "quantization": "not reported",
              "context": 1000000,
              "input": 2,
              "output": 10,
              "cacheRead": 0.2,
              "uptime1d": 99.99
            },
            {
              "model": "Claude Sonnet 5.5",
              "provider": "Claude Platform on AWS",
              "tier": "standard",
              "quantization": "not reported",
              "context": 1000000,
              "input": 2,
              "output": 10,
              "cacheRead": 0.2,
              "uptime1d": 99.98
            },
            {
              "model": "Claude Sonnet 5.5",
              "provider": "Anthropic",
              "tier": "standard",
              "quantization": "not reported",
              "context": 1000000,
              "input": 2,
              "output": 10,
              "cacheRead": 0.2,
              "uptime1d": 99.98
            },
            {
              "model": "Claude Sonnet 5.5",
              "provider": "Google Vertex",
              "tier": "regional",
              "quantization": "not reported",
              "context": 1000000,
              "input": 2.2,
              "output": 11,
              "cacheRead": 0.22,
              "uptime1d": 99.99
            },
            {
              "model": "Claude Sonnet 5.5",
              "provider": "Azure",
              "tier": "regional",
              "quantization": "not reported",
              "context": 1000000,
              "input": 2.2,
              "output": 11,
              "cacheRead": 0.22,
              "uptime1d": 100
            },
            {
              "model": "Claude Sonnet 5.5",
              "provider": "Google Vertex",
              "tier": "regional",
              "quantization": "not reported",
              "context": 1000000,
              "input": 2.2,
              "output": 11,
              "cacheRead": 0.22,
              "uptime1d": 100
            },
            {
              "model": "DeepSeek V4 Flash 0423",
              "provider": "Relace",
              "tier": "standard",
              "quantization": "fp4",
              "context": 1048576,
              "input": 0.012,
              "output": 1.28,
              "cacheRead": 0.012,
              "uptime1d": 99.92
            },
            {
              "model": "DeepSeek V4 Flash 0423",
              "provider": "OpenInference",
              "tier": "standard",
              "quantization": "fp4",
              "context": 1048576,
              "input": 0.0132,
              "output": 1.408,
              "cacheRead": 0.0132,
              "uptime1d": 99.08
            },
            {
              "model": "DeepSeek V4 Flash 0423",
              "provider": "StreamLake",
              "tier": "standard",
              "quantization": "fp8",
              "context": 1024000,
              "input": 0.042,
              "output": 0.084,
              "cacheRead": 0.0084,
              "uptime1d": 99.12
            },
            {
              "model": "DeepSeek V4 Flash 0423",
              "provider": "DeepInfra",
              "tier": "standard",
              "quantization": "fp8",
              "context": 1048576,
              "input": 0.09,
              "output": 0.18,
              "cacheRead": 0.018,
              "uptime1d": 99.67
            },
            {
              "model": "DeepSeek V4 Flash 0423",
              "provider": "GMICloud",
              "tier": "standard",
              "quantization": "fp8",
              "context": 1048575,
              "input": 0.091,
              "output": 0.182,
              "cacheRead": 0.0182,
              "uptime1d": 99.98
            },
            {
              "model": "DeepSeek V4 Flash 0423",
              "provider": "Venice",
              "tier": "standard",
              "quantization": "not reported",
              "context": 1000000,
              "input": 0.0966,
              "output": 0.1925,
              "cacheRead": 0.0196,
              "uptime1d": 96.14
            },
            {
              "model": "DeepSeek V4 Flash 0423",
              "provider": "DigitalOcean",
              "tier": "standard",
              "quantization": "not reported",
              "context": 1048576,
              "input": 0.098,
              "output": 0.196,
              "cacheRead": 0.0196,
              "uptime1d": 99.75
            },
            {
              "model": "DeepSeek V4 Flash 0423",
              "provider": "SiliconFlow",
              "tier": "standard",
              "quantization": "fp8",
              "context": 1048576,
              "input": 0.13,
              "output": 0.28,
              "cacheRead": 0.028,
              "uptime1d": 99.54
            },
            {
              "model": "DeepSeek V4 Flash 0423",
              "provider": "Alibaba",
              "tier": "standard",
              "quantization": "fp8",
              "context": 1000000,
              "input": 0.134,
              "output": 0.268,
              "cacheRead": 0.0268,
              "uptime1d": 99.21
            },
            {
              "model": "DeepSeek V4 Flash 0423",
              "provider": "Baidu",
              "tier": "standard",
              "quantization": "fp8",
              "context": 1048576,
              "input": 0.14,
              "output": 0.28,
              "cacheRead": 0.028,
              "uptime1d": 99.41
            },
            {
              "model": "DeepSeek V4 Flash 0423",
              "provider": "Novita",
              "tier": "standard",
              "quantization": "fp8",
              "context": 1048576,
              "input": 0.14,
              "output": 0.28,
              "cacheRead": 0.028,
              "uptime1d": 99.87
            },
            {
              "model": "DeepSeek V4 Flash 0423",
              "provider": "AtlasCloud",
              "tier": "standard",
              "quantization": "fp4",
              "context": 1048576,
              "input": 0.14,
              "output": 0.28,
              "cacheRead": 0.028,
              "uptime1d": 99.09
            },
            {
              "model": "DeepSeek V4 Flash 0423",
              "provider": "Parasail",
              "tier": "standard",
              "quantization": "fp8",
              "context": 1048576,
              "input": 0.14,
              "output": 0.28,
              "cacheRead": 0.07,
              "uptime1d": 99.68
            },
            {
              "model": "DeepSeek V4 Flash 0423",
              "provider": "Mancer 2",
              "tier": "standard",
              "quantization": "fp8",
              "context": 1048576,
              "input": 0.19,
              "output": 0.5,
              "cacheRead": null,
              "uptime1d": 94.91
            },
            {
              "model": "DeepSeek V4 Flash 0423",
              "provider": "Azure",
              "tier": "regional",
              "quantization": "not reported",
              "context": 1048576,
              "input": 0.21,
              "output": 0.56,
              "cacheRead": 0.031,
              "uptime1d": 95.28
            },
            {
              "model": "DeepSeek V4 Flash 0423",
              "provider": "Cloudflare",
              "tier": "standard",
              "quantization": "not reported",
              "context": 384000,
              "input": 0.44,
              "output": 1.32,
              "cacheRead": 0.014,
              "uptime1d": 98.13
            },
            {
              "model": "DeepSeek V4 Pro 0423",
              "provider": "Relace",
              "tier": "standard",
              "quantization": "fp4",
              "context": 1048576,
              "input": 0.2067,
              "output": 4.2,
              "cacheRead": 0.21,
              "uptime1d": 99.9
            },
            {
              "model": "DeepSeek V4 Pro 0423",
              "provider": "StreamLake",
              "tier": "standard",
              "quantization": "fp8",
              "context": 1024000,
              "input": 0.2088,
              "output": 0.4176,
              "cacheRead": 0.0174,
              "uptime1d": 98.67
            },
            {
              "model": "DeepSeek V4 Pro 0423",
              "provider": "Parasail",
              "tier": "standard",
              "quantization": "fp8",
              "context": 1048576,
              "input": 0.45,
              "output": 3.48,
              "cacheRead": 0.1,
              "uptime1d": 98.3
            },
            {
              "model": "DeepSeek V4 Pro 0423",
              "provider": "Reka",
              "tier": "standard",
              "quantization": "not reported",
              "context": 1048576,
              "input": 0.9,
              "output": 9,
              "cacheRead": 0.18,
              "uptime1d": 98.83
            },
            {
              "model": "DeepSeek V4 Pro 0423",
              "provider": "GMICloud",
              "tier": "standard",
              "quantization": "fp8",
              "context": 1048576,
              "input": 0.957,
              "output": 1.914,
              "cacheRead": 0.07975,
              "uptime1d": 97.05
            },
            {
              "model": "DeepSeek V4 Pro 0423",
              "provider": "DigitalOcean",
              "tier": "standard",
              "quantization": "not reported",
              "context": 1048576,
              "input": 1.044,
              "output": 2.088,
              "cacheRead": 0.2088,
              "uptime1d": 99.54
            },
            {
              "model": "DeepSeek V4 Pro 0423",
              "provider": "Cloudflare",
              "tier": "standard",
              "quantization": "not reported",
              "context": 1048576,
              "input": 1.15,
              "output": 2.55,
              "cacheRead": 0.2,
              "uptime1d": 98.29
            },
            {
              "model": "DeepSeek V4 Pro 0423",
              "provider": "DeepInfra",
              "tier": "standard",
              "quantization": "fp8",
              "context": 1048576,
              "input": 1.3,
              "output": 2.6,
              "cacheRead": 0.1,
              "uptime1d": 99.81
            },
            {
              "model": "DeepSeek V4 Pro 0423",
              "provider": "Alibaba",
              "tier": "standard",
              "quantization": "fp8",
              "context": 1000000,
              "input": 1.416,
              "output": 2.832,
              "cacheRead": 0.118,
              "uptime1d": 90.11
            },
            {
              "model": "DeepSeek V4 Pro 0423",
              "provider": "SiliconFlow",
              "tier": "standard",
              "quantization": "fp8",
              "context": 1048576,
              "input": 1.50162,
              "output": 3.135,
              "cacheRead": 0.135,
              "uptime1d": 99.21
            },
            {
              "model": "DeepSeek V4 Pro 0423",
              "provider": "Novita",
              "tier": "standard",
              "quantization": "fp8",
              "context": 1048576,
              "input": 1.6,
              "output": 3.2,
              "cacheRead": 0.135,
              "uptime1d": 99.94
            },
            {
              "model": "DeepSeek V4 Pro 0423",
              "provider": "Venice",
              "tier": "standard",
              "quantization": "not reported",
              "context": 1000000,
              "input": 1.65,
              "output": 3.301,
              "cacheRead": 0.33,
              "uptime1d": 97.05
            },
            {
              "model": "DeepSeek V4 Pro 0423",
              "provider": "AtlasCloud",
              "tier": "standard",
              "quantization": "fp4",
              "context": 1048576,
              "input": 1.68,
              "output": 3.38,
              "cacheRead": 0.13,
              "uptime1d": 99.01
            },
            {
              "model": "DeepSeek V4 Pro 0423",
              "provider": "Baidu",
              "tier": "standard",
              "quantization": "fp8",
              "context": 1048576,
              "input": 1.69,
              "output": 3.38,
              "cacheRead": 0.14,
              "uptime1d": 99.96
            },
            {
              "model": "DeepSeek V4 Pro 0423",
              "provider": "NextBit",
              "tier": "standard",
              "quantization": "fp8",
              "context": 1048576,
              "input": 1.74,
              "output": 3.48,
              "cacheRead": 0.145,
              "uptime1d": 98.71
            },
            {
              "model": "DeepSeek V4 Pro 0423",
              "provider": "Azure",
              "tier": "regional",
              "quantization": "not reported",
              "context": 1048576,
              "input": 1.91,
              "output": 3.83,
              "cacheRead": 0.16,
              "uptime1d": 99.2
            },
            {
              "model": "Gemini 3.1 Pro Preview",
              "provider": "Google Vertex",
              "tier": "flex",
              "quantization": "not reported",
              "context": 1048576,
              "input": 1,
              "output": 6,
              "cacheRead": 0.1,
              "uptime1d": 93.19
            },
            {
              "model": "Gemini 3.1 Pro Preview",
              "provider": "Google AI Studio",
              "tier": "flex",
              "quantization": "not reported",
              "context": 1048576,
              "input": 1,
              "output": 6,
              "cacheRead": 0.1,
              "uptime1d": 99.97
            },
            {
              "model": "Gemini 3.1 Pro Preview",
              "provider": "Google Vertex",
              "tier": "standard",
              "quantization": "not reported",
              "context": 1048576,
              "input": 2,
              "output": 12,
              "cacheRead": 0.2,
              "uptime1d": 98
            },
            {
              "model": "Gemini 3.1 Pro Preview",
              "provider": "Google AI Studio",
              "tier": "standard",
              "quantization": "not reported",
              "context": 1048576,
              "input": 2,
              "output": 12,
              "cacheRead": 0.2,
              "uptime1d": 99.81
            },
            {
              "model": "Gemini 3.1 Pro Preview",
              "provider": "Google Vertex",
              "tier": "priority",
              "quantization": "not reported",
              "context": 1048576,
              "input": 3.6,
              "output": 21.6,
              "cacheRead": 0.36,
              "uptime1d": 99.87
            },
            {
              "model": "Gemini 3.1 Pro Preview",
              "provider": "Google AI Studio",
              "tier": "priority",
              "quantization": "not reported",
              "context": 1048576,
              "input": 3.6,
              "output": 21.6,
              "cacheRead": 0.36,
              "uptime1d": 98.9
            },
            {
              "model": "Gemini 3.5 Flash",
              "provider": "Google Vertex",
              "tier": "flex",
              "quantization": "not reported",
              "context": 1048576,
              "input": 0.75,
              "output": 4.5,
              "cacheRead": 0.075,
              "uptime1d": 98.79
            },
            {
              "model": "Gemini 3.5 Flash",
              "provider": "Google AI Studio",
              "tier": "flex",
              "quantization": "not reported",
              "context": 1048576,
              "input": 0.75,
              "output": 4.5,
              "cacheRead": 0.075,
              "uptime1d": 99.96
            },
            {
              "model": "Gemini 3.5 Flash",
              "provider": "Google Vertex",
              "tier": "standard",
              "quantization": "not reported",
              "context": 1048576,
              "input": 1.5,
              "output": 9,
              "cacheRead": 0.15,
              "uptime1d": 98.72
            },
            {
              "model": "Gemini 3.5 Flash",
              "provider": "Google AI Studio",
              "tier": "standard",
              "quantization": "not reported",
              "context": 1048576,
              "input": 1.5,
              "output": 9,
              "cacheRead": 0.15,
              "uptime1d": 99.9
            },
            {
              "model": "Gemini 3.5 Flash",
              "provider": "Google Vertex",
              "tier": "regional",
              "quantization": "not reported",
              "context": 1048576,
              "input": 1.65,
              "output": 9.9,
              "cacheRead": 0.165,
              "uptime1d": null
            },
            {
              "model": "Gemini 3.5 Flash",
              "provider": "Google Vertex",
              "tier": "priority",
              "quantization": "not reported",
              "context": 1048576,
              "input": 2.7,
              "output": 16.2,
              "cacheRead": 0.27,
              "uptime1d": 99.9
            },
            {
              "model": "Gemini 3.5 Flash",
              "provider": "Google AI Studio",
              "tier": "priority",
              "quantization": "not reported",
              "context": 1048576,
              "input": 2.7,
              "output": 16.2,
              "cacheRead": 0.27,
              "uptime1d": 99.93
            },
            {
              "model": "Gemini 3.5 Flash Lite",
              "provider": "Google Vertex",
              "tier": "flex",
              "quantization": "not reported",
              "context": 1048576,
              "input": 0.15,
              "output": 1.25,
              "cacheRead": 0.015,
              "uptime1d": 99.99
            },
            {
              "model": "Gemini 3.5 Flash Lite",
              "provider": "Google AI Studio",
              "tier": "flex",
              "quantization": "not reported",
              "context": 1048576,
              "input": 0.15,
              "output": 1.25,
              "cacheRead": 0.015,
              "uptime1d": 99.98
            },
            {
              "model": "Gemini 3.5 Flash Lite",
              "provider": "Google AI Studio",
              "tier": "standard",
              "quantization": "not reported",
              "context": 1048576,
              "input": 0.3,
              "output": 2.5,
              "cacheRead": 0.03,
              "uptime1d": 99.95
            },
            {
              "model": "Gemini 3.5 Flash Lite",
              "provider": "Google Vertex",
              "tier": "standard",
              "quantization": "not reported",
              "context": 1048576,
              "input": 0.3,
              "output": 2.5,
              "cacheRead": 0.03,
              "uptime1d": 99.95
            },
            {
              "model": "Gemini 3.5 Flash Lite",
              "provider": "Google Vertex",
              "tier": "regional",
              "quantization": "not reported",
              "context": 1048576,
              "input": 0.33,
              "output": 2.75,
              "cacheRead": 0.033,
              "uptime1d": 100
            },
            {
              "model": "Gemini 3.5 Flash Lite",
              "provider": "Google Vertex",
              "tier": "regional",
              "quantization": "not reported",
              "context": 1048576,
              "input": 0.33,
              "output": 2.75,
              "cacheRead": 0.033,
              "uptime1d": 99.86
            },
            {
              "model": "Gemini 3.5 Flash Lite",
              "provider": "Google Vertex",
              "tier": "priority",
              "quantization": "not reported",
              "context": 1048576,
              "input": 0.54,
              "output": 4.5,
              "cacheRead": 0.054,
              "uptime1d": 99.95
            },
            {
              "model": "Gemini 3.5 Flash Lite",
              "provider": "Google AI Studio",
              "tier": "priority",
              "quantization": "not reported",
              "context": 1048576,
              "input": 0.54,
              "output": 4.5,
              "cacheRead": 0.054,
              "uptime1d": 99.96
            },
            {
              "model": "Gemini 3.8 Flash",
              "provider": "Google AI Studio",
              "tier": "flex",
              "quantization": "not reported",
              "context": 1048576,
              "input": 0.375,
              "output": 1.875,
              "cacheRead": 0.0375,
              "uptime1d": 99.93
            },
            {
              "model": "Gemini 3.8 Flash",
              "provider": "Google Vertex",
              "tier": "flex",
              "quantization": "not reported",
              "context": 1048576,
              "input": 0.375,
              "output": 1.875,
              "cacheRead": 0.0375,
              "uptime1d": 99.44
            },
            {
              "model": "Gemini 3.8 Flash",
              "provider": "Google AI Studio",
              "tier": "standard",
              "quantization": "not reported",
              "context": 1048576,
              "input": 0.75,
              "output": 3.75,
              "cacheRead": 0.075,
              "uptime1d": 99.83
            },
            {
              "model": "Gemini 3.8 Flash",
              "provider": "Google Vertex",
              "tier": "standard",
              "quantization": "not reported",
              "context": 1048576,
              "input": 0.75,
              "output": 3.75,
              "cacheRead": 0.075,
              "uptime1d": 97.49
            },
            {
              "model": "Gemini 3.8 Flash",
              "provider": "Google AI Studio",
              "tier": "priority",
              "quantization": "not reported",
              "context": 1048576,
              "input": 1.35,
              "output": 6.75,
              "cacheRead": 0.135,
              "uptime1d": 99.71
            },
            {
              "model": "Gemini 3.8 Flash",
              "provider": "Google Vertex",
              "tier": "priority",
              "quantization": "not reported",
              "context": 1048576,
              "input": 1.35,
              "output": 6.75,
              "cacheRead": 0.135,
              "uptime1d": 99.75
            },
            {
              "model": "GLM 5.3",
              "provider": "Relace",
              "tier": "standard",
              "quantization": "not reported",
              "context": 1048576,
              "input": 0.03,
              "output": 12,
              "cacheRead": 0.03,
              "uptime1d": 99.96
            },
            {
              "model": "GLM 5.3",
              "provider": "Wafer",
              "tier": "regional",
              "quantization": "not reported",
              "context": 1048576,
              "input": 0.07,
              "output": 7,
              "cacheRead": 0.065,
              "uptime1d": 99.8
            },
            {
              "model": "GLM 5.3",
              "provider": "InferenceNet",
              "tier": "standard",
              "quantization": "not reported",
              "context": 1048576,
              "input": 0.14,
              "output": 4.4,
              "cacheRead": 0.07,
              "uptime1d": 99.72
            },
            {
              "model": "GLM 5.3",
              "provider": "Wafer",
              "tier": "standard",
              "quantization": "not reported",
              "context": 1048576,
              "input": 0.15,
              "output": 7,
              "cacheRead": 0.14,
              "uptime1d": 99.6
            },
            {
              "model": "GLM 5.3",
              "provider": "Reka",
              "tier": "standard",
              "quantization": "not reported",
              "context": 262144,
              "input": 0.17,
              "output": 3,
              "cacheRead": 0.169,
              "uptime1d": 99.76
            },
            {
              "model": "GLM 5.3",
              "provider": "Morph",
              "tier": "standard",
              "quantization": "fp8",
              "context": 1048576,
              "input": 0.179,
              "output": 3.553,
              "cacheRead": 0.137,
              "uptime1d": 98.66
            },
            {
              "model": "GLM 5.3",
              "provider": "Makora",
              "tier": "standard",
              "quantization": "fp4",
              "context": 980000,
              "input": 0.18,
              "output": 4.4,
              "cacheRead": 0.19,
              "uptime1d": 96.34
            },
            {
              "model": "GLM 5.3",
              "provider": "AkashML",
              "tier": "standard",
              "quantization": "fp8",
              "context": 1048576,
              "input": 0.19,
              "output": 4.4,
              "cacheRead": 0.19,
              "uptime1d": 99.93
            },
            {
              "model": "GLM 5.3",
              "provider": "Sail Research",
              "tier": "regional",
              "quantization": "fp8",
              "context": 1048576,
              "input": 0.2,
              "output": 3.4,
              "cacheRead": 0.15,
              "uptime1d": 99.48
            },
            {
              "model": "GLM 5.3",
              "provider": "Sail Research",
              "tier": "standard",
              "quantization": "fp8",
              "context": 1048576,
              "input": 0.2,
              "output": 3.4,
              "cacheRead": 0.15,
              "uptime1d": 99.09
            },
            {
              "model": "GLM 5.3",
              "provider": "Novita",
              "tier": "standard",
              "quantization": "fp8",
              "context": 1048576,
              "input": 0.42,
              "output": 1.32,
              "cacheRead": 0.078,
              "uptime1d": 97.23
            },
            {
              "model": "GLM 5.3",
              "provider": "DeepInfra",
              "tier": "standard",
              "quantization": "fp4",
              "context": 1048576,
              "input": 0.5625,
              "output": 2.5,
              "cacheRead": 0.125,
              "uptime1d": 97.25
            },
            {
              "model": "GLM 5.3",
              "provider": "Inceptron",
              "tier": "standard",
              "quantization": "fp4",
              "context": 1048576,
              "input": 0.6,
              "output": 3.39,
              "cacheRead": 0.2,
              "uptime1d": 98.29
            },
            {
              "model": "GLM 5.3",
              "provider": "SiliconFlow",
              "tier": "standard",
              "quantization": "fp8",
              "context": 1048576,
              "input": 0.7,
              "output": 2.2,
              "cacheRead": 0.13,
              "uptime1d": 99.84
            },
            {
              "model": "GLM 5.3",
              "provider": "Phala",
              "tier": "standard",
              "quantization": "not reported",
              "context": 1048576,
              "input": 0.84,
              "output": 2.64,
              "cacheRead": 0.156,
              "uptime1d": 99.16
            },
            {
              "model": "GLM 5.3",
              "provider": "DigitalOcean",
              "tier": "standard",
              "quantization": "not reported",
              "context": 1048576,
              "input": 0.91,
              "output": 2.86,
              "cacheRead": 0.169,
              "uptime1d": 99.8
            },
            {
              "model": "GLM 5.3",
              "provider": "GMICloud",
              "tier": "standard",
              "quantization": "fp8",
              "context": 1048576,
              "input": 0.98,
              "output": 3.08,
              "cacheRead": 0.182,
              "uptime1d": 98.75
            },
            {
              "model": "GLM 5.3",
              "provider": "Alibaba",
              "tier": "standard",
              "quantization": "not reported",
              "context": 1000000,
              "input": 1.19,
              "output": 3.74,
              "cacheRead": 0.238,
              "uptime1d": 99.95
            },
            {
              "model": "GLM 5.3",
              "provider": "Decart",
              "tier": "standard",
              "quantization": "fp4",
              "context": 1048576,
              "input": 1.19,
              "output": 3.74,
              "cacheRead": 0.1955,
              "uptime1d": 99.99
            },
            {
              "model": "GLM 5.3",
              "provider": "Friendli",
              "tier": "standard",
              "quantization": "not reported",
              "context": 1048576,
              "input": 1.26,
              "output": 3.96,
              "cacheRead": 0.234,
              "uptime1d": 99.94
            },
            {
              "model": "GLM 5.3",
              "provider": "Mistral",
              "tier": "standard",
              "quantization": "nvfp4",
              "context": 1048576,
              "input": 1.4,
              "output": 4.4,
              "cacheRead": 0.14,
              "uptime1d": 99.73
            },
            {
              "model": "GLM 5.3",
              "provider": "Baidu",
              "tier": "standard",
              "quantization": "fp8",
              "context": 1048576,
              "input": 1.4,
              "output": 4.4,
              "cacheRead": 0.26,
              "uptime1d": 99.89
            },
            {
              "model": "GLM 5.3",
              "provider": "BaseTen",
              "tier": "standard",
              "quantization": "fp4",
              "context": 1048576,
              "input": 1.4,
              "output": 4.4,
              "cacheRead": 0.14,
              "uptime1d": 97.75
            },
            {
              "model": "GLM 5.3",
              "provider": "Mistral",
              "tier": "standard",
              "quantization": "nvfp4",
              "context": 1048576,
              "input": 1.4,
              "output": 4.4,
              "cacheRead": 0.14,
              "uptime1d": 99.34
            },
            {
              "model": "GLM 5.3",
              "provider": "Nebius",
              "tier": "standard",
              "quantization": "fp4",
              "context": 1024000,
              "input": 1.4,
              "output": 4.4,
              "cacheRead": null,
              "uptime1d": 96.05
            },
            {
              "model": "GLM 5.3",
              "provider": "Crusoe",
              "tier": "standard",
              "quantization": "fp4",
              "context": 1048576,
              "input": 1.4,
              "output": 4.4,
              "cacheRead": 0.26,
              "uptime1d": 98.48
            },
            {
              "model": "GLM 5.3",
              "provider": "PrimeIntellect",
              "tier": "standard",
              "quantization": "not reported",
              "context": 1048576,
              "input": 1.4,
              "output": 4.4,
              "cacheRead": 0.26,
              "uptime1d": 99.47
            },
            {
              "model": "GLM 5.3",
              "provider": "Venice",
              "tier": "standard",
              "quantization": "not reported",
              "context": 1000000,
              "input": 1.4,
              "output": 4.4,
              "cacheRead": 0.26,
              "uptime1d": 98.63
            },
            {
              "model": "GLM 5.3",
              "provider": "Together",
              "tier": "standard",
              "quantization": "not reported",
              "context": 1048575,
              "input": 1.4,
              "output": 4.4,
              "cacheRead": 0.26,
              "uptime1d": 96.37
            },
            {
              "model": "GLM 5.3",
              "provider": "Parasail",
              "tier": "standard",
              "quantization": "fp8",
              "context": 1048576,
              "input": 1.4,
              "output": 4.4,
              "cacheRead": 0.26,
              "uptime1d": 99.7
            },
            {
              "model": "GLM 5.3",
              "provider": "Modal",
              "tier": "standard",
              "quantization": "not reported",
              "context": 1048576,
              "input": 1.4,
              "output": 4.4,
              "cacheRead": 0.26,
              "uptime1d": 98.13
            },
            {
              "model": "GLM 5.3",
              "provider": "BaseTen",
              "tier": "standard",
              "quantization": "fp4",
              "context": 1048576,
              "input": 1.4,
              "output": 4.4,
              "cacheRead": 0.14,
              "uptime1d": 95.1
            },
            {
              "model": "GLM 5.3",
              "provider": "Fireworks",
              "tier": "standard",
              "quantization": "not reported",
              "context": 1048576,
              "input": 1.4,
              "output": 4.4,
              "cacheRead": 0.26,
              "uptime1d": 99.5
            },
            {
              "model": "GLM 5.3",
              "provider": "Cloudflare",
              "tier": "standard",
              "quantization": "not reported",
              "context": 1048576,
              "input": 1.4,
              "output": 4.4,
              "cacheRead": 0.26,
              "uptime1d": 98.18
            },
            {
              "model": "GLM 5.3",
              "provider": "AtlasCloud",
              "tier": "standard",
              "quantization": "fp8",
              "context": 1048576,
              "input": 1.4,
              "output": 4.4,
              "cacheRead": 0.26,
              "uptime1d": 99.94
            },
            {
              "model": "GLM 5.3",
              "provider": "Z.AI",
              "tier": "standard",
              "quantization": "fp8",
              "context": 1048576,
              "input": 1.4,
              "output": 4.4,
              "cacheRead": 0.26,
              "uptime1d": 99.84
            },
            {
              "model": "GLM 5.3",
              "provider": "Mistral",
              "tier": "standard",
              "quantization": "nvfp4",
              "context": 1048576,
              "input": 1.54,
              "output": 4.84,
              "cacheRead": 0.154,
              "uptime1d": 99.82
            },
            {
              "model": "GLM 5.3",
              "provider": "Fireworks",
              "tier": "fast",
              "quantization": "not reported",
              "context": 1048576,
              "input": 2.1,
              "output": 6.6,
              "cacheRead": 0.39,
              "uptime1d": 99.59
            },
            {
              "model": "GLM 5.3",
              "provider": "BaseTen",
              "tier": "fast",
              "quantization": "fp8",
              "context": 1048576,
              "input": 2.1,
              "output": 6.6,
              "cacheRead": 0.21,
              "uptime1d": 99.65
            },
            {
              "model": "GLM 5.3",
              "provider": "BaseTen",
              "tier": "fast",
              "quantization": "fp8",
              "context": 1048576,
              "input": 2.1,
              "output": 6.6,
              "cacheRead": 0.21,
              "uptime1d": 99.21
            },
            {
              "model": "GLM 5.3",
              "provider": "Alibaba",
              "tier": "fast",
              "quantization": "not reported",
              "context": 1000000,
              "input": 2.8,
              "output": 8.8,
              "cacheRead": 0.56,
              "uptime1d": 100
            },
            {
              "model": "GPT-5.5",
              "provider": "OpenAI",
              "tier": "flex",
              "quantization": "not reported",
              "context": 1050000,
              "input": 2.5,
              "output": 15,
              "cacheRead": 0.25,
              "uptime1d": 100
            },
            {
              "model": "GPT-5.5",
              "provider": "Azure",
              "tier": "standard",
              "quantization": "not reported",
              "context": 1050000,
              "input": 5,
              "output": 30,
              "cacheRead": 0.5,
              "uptime1d": 99.96
            },
            {
              "model": "GPT-5.5",
              "provider": "OpenAI",
              "tier": "standard",
              "quantization": "not reported",
              "context": 1050000,
              "input": 5,
              "output": 30,
              "cacheRead": 0.5,
              "uptime1d": 99.99
            },
            {
              "model": "GPT-5.5",
              "provider": "Azure",
              "tier": "regional",
              "quantization": "not reported",
              "context": 1050000,
              "input": 5.5,
              "output": 33,
              "cacheRead": 0.55,
              "uptime1d": 100
            },
            {
              "model": "GPT-5.5",
              "provider": "Azure",
              "tier": "regional",
              "quantization": "not reported",
              "context": 1050000,
              "input": 5.5,
              "output": 33,
              "cacheRead": 0.55,
              "uptime1d": 100
            },
            {
              "model": "GPT-5.5",
              "provider": "Amazon Bedrock",
              "tier": "regional",
              "quantization": "not reported",
              "context": 1050000,
              "input": 5.5,
              "output": 33,
              "cacheRead": 0.55,
              "uptime1d": null
            },
            {
              "model": "GPT-5.5",
              "provider": "OpenAI",
              "tier": "fast",
              "quantization": "not reported",
              "context": 1050000,
              "input": 12.5,
              "output": 75,
              "cacheRead": 1.25,
              "uptime1d": 100
            },
            {
              "model": "GPT-6 Astra",
              "provider": "OpenAI",
              "tier": "flex",
              "quantization": "not reported",
              "context": 1050000,
              "input": 5,
              "output": 25,
              "cacheRead": 0.5,
              "uptime1d": 100
            },
            {
              "model": "GPT-6 Astra",
              "provider": "Azure",
              "tier": "standard",
              "quantization": "not reported",
              "context": 1050000,
              "input": 10,
              "output": 50,
              "cacheRead": 1,
              "uptime1d": 99.97
            },
            {
              "model": "GPT-6 Astra",
              "provider": "OpenAI",
              "tier": "standard",
              "quantization": "not reported",
              "context": 1050000,
              "input": 10,
              "output": 50,
              "cacheRead": 1,
              "uptime1d": 99.99
            },
            {
              "model": "GPT-6 Astra",
              "provider": "Amazon Bedrock",
              "tier": "regional",
              "quantization": "not reported",
              "context": 1050000,
              "input": 11,
              "output": 55,
              "cacheRead": 1.1,
              "uptime1d": null
            },
            {
              "model": "GPT-6 Astra",
              "provider": "Azure",
              "tier": "regional",
              "quantization": "not reported",
              "context": 1050000,
              "input": 11,
              "output": 55,
              "cacheRead": 1.1,
              "uptime1d": 100
            },
            {
              "model": "GPT-6 Astra",
              "provider": "OpenAI",
              "tier": "fast",
              "quantization": "not reported",
              "context": 1050000,
              "input": 20,
              "output": 100,
              "cacheRead": 2,
              "uptime1d": 99.97
            },
            {
              "model": "GPT-6 Astra",
              "provider": "OpenAI",
              "tier": "ultrafast",
              "quantization": "not reported",
              "context": 1050000,
              "input": 60,
              "output": 300,
              "cacheRead": 6,
              "uptime1d": 100
            },
            {
              "model": "GPT-6 Luna",
              "provider": "OpenAI",
              "tier": "flex",
              "quantization": "not reported",
              "context": 1050000,
              "input": 0.05,
              "output": 0.25,
              "cacheRead": 0.005,
              "uptime1d": 97.74
            },
            {
              "model": "GPT-6 Luna",
              "provider": "OpenAI",
              "tier": "standard",
              "quantization": "not reported",
              "context": 1050000,
              "input": 0.1,
              "output": 0.5,
              "cacheRead": 0.01,
              "uptime1d": 99.99
            },
            {
              "model": "GPT-6 Luna",
              "provider": "Azure",
              "tier": "standard",
              "quantization": "not reported",
              "context": 1050000,
              "input": 0.1,
              "output": 0.5,
              "cacheRead": 0.01,
              "uptime1d": 99.92
            },
            {
              "model": "GPT-6 Luna",
              "provider": "Azure",
              "tier": "regional",
              "quantization": "not reported",
              "context": 1050000,
              "input": 0.11,
              "output": 0.55,
              "cacheRead": 0.011,
              "uptime1d": 99.99
            },
            {
              "model": "GPT-6 Luna",
              "provider": "Azure",
              "tier": "regional",
              "quantization": "not reported",
              "context": 1050000,
              "input": 0.11,
              "output": 0.55,
              "cacheRead": 0.011,
              "uptime1d": 99.98
            },
            {
              "model": "GPT-6 Luna",
              "provider": "Amazon Bedrock",
              "tier": "regional",
              "quantization": "not reported",
              "context": 1050000,
              "input": 0.11,
              "output": 0.55,
              "cacheRead": 0.011,
              "uptime1d": 98.8
            },
            {
              "model": "GPT-6 Luna",
              "provider": "OpenAI",
              "tier": "fast",
              "quantization": "not reported",
              "context": 1050000,
              "input": 0.2,
              "output": 1,
              "cacheRead": 0.02,
              "uptime1d": 99.99
            },
            {
              "model": "GPT-6 Sol",
              "provider": "OpenAI",
              "tier": "flex",
              "quantization": "not reported",
              "context": 1050000,
              "input": 1,
              "output": 5,
              "cacheRead": 0.1,
              "uptime1d": 99.99
            },
            {
              "model": "GPT-6 Sol",
              "provider": "OpenAI",
              "tier": "standard",
              "quantization": "not reported",
              "context": 1050000,
              "input": 2,
              "output": 10,
              "cacheRead": 0.2,
              "uptime1d": 100
            },
            {
              "model": "GPT-6 Sol",
              "provider": "Azure",
              "tier": "standard",
              "quantization": "not reported",
              "context": 1050000,
              "input": 2,
              "output": 10,
              "cacheRead": 0.2,
              "uptime1d": 99.99
            },
            {
              "model": "GPT-6 Sol",
              "provider": "Azure",
              "tier": "regional",
              "quantization": "not reported",
              "context": 1050000,
              "input": 2.2,
              "output": 11,
              "cacheRead": 0.22,
              "uptime1d": 99.98
            },
            {
              "model": "GPT-6 Sol",
              "provider": "Azure",
              "tier": "regional",
              "quantization": "not reported",
              "context": 1050000,
              "input": 2.2,
              "output": 11,
              "cacheRead": 0.22,
              "uptime1d": 100
            },
            {
              "model": "GPT-6 Sol",
              "provider": "Amazon Bedrock",
              "tier": "regional",
              "quantization": "not reported",
              "context": 1050000,
              "input": 2.2,
              "output": 11,
              "cacheRead": 0.22,
              "uptime1d": 100
            },
            {
              "model": "GPT-6 Sol",
              "provider": "OpenAI",
              "tier": "fast",
              "quantization": "not reported",
              "context": 1050000,
              "input": 4,
              "output": 20,
              "cacheRead": 0.4,
              "uptime1d": 99.99
            },
            {
              "model": "gpt-oss-120b",
              "provider": "CoreWeave",
              "tier": "standard",
              "quantization": "fp4",
              "context": 131072,
              "input": 0.03,
              "output": 0.17,
              "cacheRead": 0.03,
              "uptime1d": 98.78
            },
            {
              "model": "gpt-oss-120b",
              "provider": "DekaLLM",
              "tier": "standard",
              "quantization": "bf16",
              "context": 131072,
              "input": 0.03,
              "output": 0.18,
              "cacheRead": 0.03,
              "uptime1d": 99.57
            },
            {
              "model": "gpt-oss-120b",
              "provider": "DeepInfra",
              "tier": "standard",
              "quantization": "bf16",
              "context": 131072,
              "input": 0.037,
              "output": 0.17,
              "cacheRead": null,
              "uptime1d": 98.93
            },
            {
              "model": "gpt-oss-120b",
              "provider": "AkashML",
              "tier": "standard",
              "quantization": "bf16",
              "context": 131072,
              "input": 0.037,
              "output": 0.187,
              "cacheRead": 0.037,
              "uptime1d": 99.98
            },
            {
              "model": "gpt-oss-120b",
              "provider": "Mancer 2",
              "tier": "standard",
              "quantization": "fp8",
              "context": 131072,
              "input": 0.045,
              "output": 0.25,
              "cacheRead": null,
              "uptime1d": 98.58
            },
            {
              "model": "gpt-oss-120b",
              "provider": "Crusoe",
              "tier": "standard",
              "quantization": "bf16",
              "context": 131072,
              "input": 0.05,
              "output": 0.25,
              "cacheRead": 0.05,
              "uptime1d": 99.98
            },
            {
              "model": "gpt-oss-120b",
              "provider": "Novita",
              "tier": "standard",
              "quantization": "fp4",
              "context": 131072,
              "input": 0.05,
              "output": 0.25,
              "cacheRead": null,
              "uptime1d": 98.57
            },
            {
              "model": "gpt-oss-120b",
              "provider": "DigitalOcean",
              "tier": "standard",
              "quantization": "not reported",
              "context": 128000,
              "input": 0.06,
              "output": 0.42,
              "cacheRead": 0.012,
              "uptime1d": 99.98
            },
            {
              "model": "gpt-oss-120b",
              "provider": "Google Vertex",
              "tier": "standard",
              "quantization": "not reported",
              "context": 131072,
              "input": 0.09,
              "output": 0.36,
              "cacheRead": null,
              "uptime1d": 66.93
            },
            {
              "model": "gpt-oss-120b",
              "provider": "BaseTen",
              "tier": "standard",
              "quantization": "fp4",
              "context": 128072,
              "input": 0.1,
              "output": 0.5,
              "cacheRead": 0.1,
              "uptime1d": 99.96
            },
            {
              "model": "gpt-oss-120b",
              "provider": "BaseTen",
              "tier": "standard",
              "quantization": "fp4",
              "context": 128072,
              "input": 0.1,
              "output": 0.5,
              "cacheRead": 0.1,
              "uptime1d": 99.97
            },
            {
              "model": "gpt-oss-120b",
              "provider": "Parasail",
              "tier": "standard",
              "quantization": "fp4",
              "context": 131072,
              "input": 0.1,
              "output": 0.75,
              "cacheRead": 0.055,
              "uptime1d": 99.88
            },
            {
              "model": "gpt-oss-120b",
              "provider": "SambaNova",
              "tier": "standard",
              "quantization": "not reported",
              "context": 131072,
              "input": 0.14,
              "output": 0.95,
              "cacheRead": null,
              "uptime1d": 99.68
            },
            {
              "model": "gpt-oss-120b",
              "provider": "Amazon Bedrock",
              "tier": "regional",
              "quantization": "not reported",
              "context": 131072,
              "input": 0.15,
              "output": 0.6,
              "cacheRead": null,
              "uptime1d": 99.97
            },
            {
              "model": "gpt-oss-120b",
              "provider": "Nebius",
              "tier": "standard",
              "quantization": "fp4",
              "context": 131072,
              "input": 0.15,
              "output": 0.6,
              "cacheRead": null,
              "uptime1d": 97.3
            },
            {
              "model": "gpt-oss-120b",
              "provider": "Amazon Bedrock",
              "tier": "standard",
              "quantization": "not reported",
              "context": 131072,
              "input": 0.15,
              "output": 0.6,
              "cacheRead": null,
              "uptime1d": 99.46
            },
            {
              "model": "gpt-oss-120b",
              "provider": "DeepInfra",
              "tier": "standard",
              "quantization": "bf16",
              "context": 131072,
              "input": 0.15,
              "output": 0.6,
              "cacheRead": null,
              "uptime1d": 99.98
            },
            {
              "model": "gpt-oss-120b",
              "provider": "SiliconFlow",
              "tier": "standard",
              "quantization": "fp8",
              "context": 131072,
              "input": 0.15,
              "output": 0.6,
              "cacheRead": 0.075,
              "uptime1d": 83.35
            },
            {
              "model": "gpt-oss-120b",
              "provider": "Phala",
              "tier": "standard",
              "quantization": "not reported",
              "context": 131072,
              "input": 0.15,
              "output": 0.6,
              "cacheRead": null,
              "uptime1d": 99.27
            },
            {
              "model": "gpt-oss-120b",
              "provider": "Together",
              "tier": "standard",
              "quantization": "not reported",
              "context": 131072,
              "input": 0.15,
              "output": 0.6,
              "cacheRead": null,
              "uptime1d": 87.17
            },
            {
              "model": "gpt-oss-120b",
              "provider": "Groq",
              "tier": "standard",
              "quantization": "not reported",
              "context": 131072,
              "input": 0.15,
              "output": 0.6,
              "cacheRead": 0.075,
              "uptime1d": 99.34
            },
            {
              "model": "gpt-oss-120b",
              "provider": "Mara",
              "tier": "standard",
              "quantization": "not reported",
              "context": 131072,
              "input": 0.15,
              "output": 0.75,
              "cacheRead": null,
              "uptime1d": 96.56
            },
            {
              "model": "gpt-oss-120b",
              "provider": "Cerebras",
              "tier": "standard",
              "quantization": "fp16",
              "context": 131072,
              "input": 0.35,
              "output": 0.75,
              "cacheRead": 0.35,
              "uptime1d": 99.98
            },
            {
              "model": "Grok 4.7",
              "provider": "xAI",
              "tier": "standard",
              "quantization": "not reported",
              "context": 500000,
              "input": 2,
              "output": 6,
              "cacheRead": 0.5,
              "uptime1d": 99.46
            },
            {
              "model": "Grok 4.7",
              "provider": "xAI",
              "tier": "standard",
              "quantization": "not reported",
              "context": 500000,
              "input": 2,
              "output": 6,
              "cacheRead": 0.5,
              "uptime1d": 99.32
            },
            {
              "model": "Grok 4.7",
              "provider": "xAI",
              "tier": "regional",
              "quantization": "not reported",
              "context": 500000,
              "input": 2.2,
              "output": 6.6,
              "cacheRead": 0.55,
              "uptime1d": 100
            },
            {
              "model": "Grok 4.7",
              "provider": "xAI",
              "tier": "priority",
              "quantization": "not reported",
              "context": 500000,
              "input": 4,
              "output": 12,
              "cacheRead": 1,
              "uptime1d": 99.55
            },
            {
              "model": "Grok 4.7",
              "provider": "xAI",
              "tier": "priority",
              "quantization": "not reported",
              "context": 500000,
              "input": 4,
              "output": 12,
              "cacheRead": 1,
              "uptime1d": 99.38
            },
            {
              "model": "Kimi K3",
              "provider": "Relace",
              "tier": "standard",
              "quantization": "fp4",
              "context": 1048576,
              "input": 0.83,
              "output": 13,
              "cacheRead": 0.45,
              "uptime1d": 99.66
            },
            {
              "model": "Kimi K3",
              "provider": "Sail Research",
              "tier": "standard",
              "quantization": "fp4",
              "context": 1048576,
              "input": 0.84,
              "output": 13.5,
              "cacheRead": 0.3,
              "uptime1d": 99.9
            },
            {
              "model": "Kimi K3",
              "provider": "InferenceNet",
              "tier": "standard",
              "quantization": "fp4",
              "context": 1048576,
              "input": 0.95,
              "output": 14,
              "cacheRead": 0.31,
              "uptime1d": 99.9
            },
            {
              "model": "Kimi K3",
              "provider": "Wafer",
              "tier": "standard",
              "quantization": "not reported",
              "context": 1048576,
              "input": 0.95,
              "output": 14,
              "cacheRead": 0.4,
              "uptime1d": 99.48
            },
            {
              "model": "Kimi K3",
              "provider": "Morph",
              "tier": "standard",
              "quantization": "fp8",
              "context": 1048576,
              "input": 1.274,
              "output": 13.296,
              "cacheRead": 0.278,
              "uptime1d": 99.56
            },
            {
              "model": "Kimi K3",
              "provider": "AkashML",
              "tier": "standard",
              "quantization": "fp4",
              "context": 1048576,
              "input": 1.3,
              "output": 14,
              "cacheRead": 1.3,
              "uptime1d": 98.42
            },
            {
              "model": "Kimi K3",
              "provider": "Makora",
              "tier": "standard",
              "quantization": "not reported",
              "context": 1048576,
              "input": 1.53,
              "output": 12.75,
              "cacheRead": 0.204,
              "uptime1d": 96.99
            },
            {
              "model": "Kimi K3",
              "provider": "Phala",
              "tier": "standard",
              "quantization": "not reported",
              "context": 1048576,
              "input": 1.95,
              "output": 9.75,
              "cacheRead": 0.195,
              "uptime1d": 97.01
            },
            {
              "model": "Kimi K3",
              "provider": "Decart",
              "tier": "standard",
              "quantization": "mxfp4",
              "context": 1048576,
              "input": 2.01,
              "output": 10.05,
              "cacheRead": 0.201,
              "uptime1d": 86.93
            },
            {
              "model": "Kimi K3",
              "provider": "DigitalOcean",
              "tier": "standard",
              "quantization": "not reported",
              "context": 1048576,
              "input": 2.55,
              "output": 12.95,
              "cacheRead": 0.255,
              "uptime1d": 99.92
            },
            {
              "model": "Kimi K3",
              "provider": "Together",
              "tier": "standard",
              "quantization": "not reported",
              "context": 1048576,
              "input": 2.7,
              "output": 13.5,
              "cacheRead": 0.27,
              "uptime1d": 99.32
            },
            {
              "model": "Kimi K3",
              "provider": "Wafer",
              "tier": "regional",
              "quantization": "not reported",
              "context": 1048576,
              "input": 2.8,
              "output": 14,
              "cacheRead": 0.3,
              "uptime1d": 98.97
            },
            {
              "model": "Kimi K3",
              "provider": "DeepInfra",
              "tier": "standard",
              "quantization": "mxfp4",
              "context": 1048576,
              "input": 2.85,
              "output": 14.25,
              "cacheRead": 0.285,
              "uptime1d": 99.36
            },
            {
              "model": "Kimi K3",
              "provider": "Amazon Bedrock",
              "tier": "regional",
              "quantization": "not reported",
              "context": 1048576,
              "input": 3,
              "output": 15,
              "cacheRead": 0.3,
              "uptime1d": 93.2
            },
            {
              "model": "Kimi K3",
              "provider": "Chutes",
              "tier": "standard",
              "quantization": "mxfp4",
              "context": 1048576,
              "input": 3,
              "output": 15,
              "cacheRead": 0.3,
              "uptime1d": 97.22
            },
            {
              "model": "Kimi K3",
              "provider": "Parasail",
              "tier": "standard",
              "quantization": "fp4",
              "context": 1048576,
              "input": 3,
              "output": 15,
              "cacheRead": 0.3,
              "uptime1d": 98.44
            },
            {
              "model": "Kimi K3",
              "provider": "Modal",
              "tier": "standard",
              "quantization": "mxfp4",
              "context": 1048576,
              "input": 3,
              "output": 15,
              "cacheRead": 0.3,
              "uptime1d": 98.25
            },
            {
              "model": "Kimi K3",
              "provider": "Fireworks",
              "tier": "standard",
              "quantization": "not reported",
              "context": 1048576,
              "input": 3,
              "output": 15,
              "cacheRead": 0.3,
              "uptime1d": 99.32
            },
            {
              "model": "Kimi K3",
              "provider": "BaseTen",
              "tier": "standard",
              "quantization": "fp8",
              "context": 1048576,
              "input": 3,
              "output": 15,
              "cacheRead": 0.3,
              "uptime1d": 98.02
            },
            {
              "model": "Kimi K3",
              "provider": "Moonshot AI",
              "tier": "standard",
              "quantization": "mxfp4",
              "context": 1048576,
              "input": 3,
              "output": 15,
              "cacheRead": 0.3,
              "uptime1d": 99.99
            },
            {
              "model": "Kimi K3",
              "provider": "Alibaba",
              "tier": "standard",
              "quantization": "not reported",
              "context": 1048576,
              "input": 3.45,
              "output": 17.25,
              "cacheRead": 0.345,
              "uptime1d": 99.19
            },
            {
              "model": "Kimi K3",
              "provider": "InferenceNet",
              "tier": "fast",
              "quantization": "fp4",
              "context": 250000,
              "input": 3.5,
              "output": 15,
              "cacheRead": 0.45,
              "uptime1d": 99.8
            },
            {
              "model": "Kimi K3",
              "provider": "Fireworks",
              "tier": "regional",
              "quantization": "not reported",
              "context": 1048576,
              "input": 4.5,
              "output": 22.5,
              "cacheRead": 0.45,
              "uptime1d": 98.97
            },
            {
              "model": "Kimi K3",
              "provider": "Fireworks",
              "tier": "fast",
              "quantization": "not reported",
              "context": 1048576,
              "input": 4.5,
              "output": 22.5,
              "cacheRead": 0.45,
              "uptime1d": 96.99
            },
            {
              "model": "Llama 3.3 70B Instruct",
              "provider": "DeepInfra",
              "tier": "standard",
              "quantization": "fp8",
              "context": 131072,
              "input": 0.1,
              "output": 0.32,
              "cacheRead": null,
              "uptime1d": 98.1
            },
            {
              "model": "Llama 3.3 70B Instruct",
              "provider": "Novita",
              "tier": "standard",
              "quantization": "bf16",
              "context": 12288,
              "input": 0.135,
              "output": 0.4,
              "cacheRead": null,
              "uptime1d": 98.08
            },
            {
              "model": "Llama 3.3 70B Instruct",
              "provider": "AkashML",
              "tier": "standard",
              "quantization": "fp8",
              "context": 131072,
              "input": 0.2,
              "output": 0.52,
              "cacheRead": 0.1,
              "uptime1d": 99.27
            },
            {
              "model": "Llama 3.3 70B Instruct",
              "provider": "Parasail",
              "tier": "standard",
              "quantization": "fp8",
              "context": 131072,
              "input": 0.22,
              "output": 0.5,
              "cacheRead": 0.11,
              "uptime1d": 99.65
            },
            {
              "model": "Llama 3.3 70B Instruct",
              "provider": "Cloudflare",
              "tier": "standard",
              "quantization": "fp8",
              "context": 24000,
              "input": 0.293,
              "output": 2.253,
              "cacheRead": null,
              "uptime1d": 98.85
            },
            {
              "model": "Llama 3.3 70B Instruct",
              "provider": "SambaNova",
              "tier": "standard",
              "quantization": "not reported",
              "context": 131072,
              "input": 0.45,
              "output": 0.9,
              "cacheRead": null,
              "uptime1d": 98.56
            },
            {
              "model": "Llama 3.3 70B Instruct",
              "provider": "Groq",
              "tier": "standard",
              "quantization": "not reported",
              "context": 131072,
              "input": 0.59,
              "output": 0.79,
              "cacheRead": 0.295,
              "uptime1d": 99.85
            },
            {
              "model": "Llama 3.3 70B Instruct",
              "provider": "CoreWeave",
              "tier": "standard",
              "quantization": "fp16",
              "context": 128000,
              "input": 0.71,
              "output": 0.71,
              "cacheRead": 0.71,
              "uptime1d": 97.72
            },
            {
              "model": "Llama 3.3 70B Instruct",
              "provider": "Google Vertex",
              "tier": "regional",
              "quantization": "not reported",
              "context": 128000,
              "input": 0.72,
              "output": 0.72,
              "cacheRead": null,
              "uptime1d": null
            },
            {
              "model": "Llama 3.3 70B Instruct",
              "provider": "Google Vertex",
              "tier": "standard",
              "quantization": "not reported",
              "context": 128000,
              "input": 0.72,
              "output": 0.72,
              "cacheRead": null,
              "uptime1d": null
            },
            {
              "model": "Llama 3.3 70B Instruct",
              "provider": "Together",
              "tier": "standard",
              "quantization": "not reported",
              "context": 131072,
              "input": 1.04,
              "output": 1.04,
              "cacheRead": null,
              "uptime1d": 93.22
            },
            {
              "model": "Llama 4 Maverick",
              "provider": "DigitalOcean",
              "tier": "standard",
              "quantization": "not reported",
              "context": 128000,
              "input": 0.1875,
              "output": 0.6525,
              "cacheRead": null,
              "uptime1d": 99.3
            },
            {
              "model": "Llama 4 Maverick",
              "provider": "Novita",
              "tier": "standard",
              "quantization": "fp8",
              "context": 1048576,
              "input": 0.27,
              "output": 0.85,
              "cacheRead": null,
              "uptime1d": 97.58
            },
            {
              "model": "Llama 4 Maverick",
              "provider": "Parasail",
              "tier": "standard",
              "quantization": "fp8",
              "context": 524288,
              "input": 0.35,
              "output": 1,
              "cacheRead": 0.17,
              "uptime1d": 99.87
            },
            {
              "model": "Llama 4 Maverick",
              "provider": "Google Vertex",
              "tier": "regional",
              "quantization": "not reported",
              "context": 524288,
              "input": 0.35,
              "output": 1.15,
              "cacheRead": null,
              "uptime1d": null
            },
            {
              "model": "Mistral Large 3 2512",
              "provider": "Mistral",
              "tier": "standard",
              "quantization": "not reported",
              "context": 262144,
              "input": 0.5,
              "output": 1.5,
              "cacheRead": 0.05,
              "uptime1d": 99.85
            },
            {
              "model": "Mistral Large 3 2512",
              "provider": "Mistral",
              "tier": "regional",
              "quantization": "not reported",
              "context": 262144,
              "input": 0.55,
              "output": 1.65,
              "cacheRead": 0.055,
              "uptime1d": 99.84
            },
            {
              "model": "Mistral Medium 3.5",
              "provider": "Mistral",
              "tier": "standard",
              "quantization": "not reported",
              "context": 262144,
              "input": 1.5,
              "output": 7.5,
              "cacheRead": null,
              "uptime1d": 99.91
            },
            {
              "model": "Mistral Medium 3.5",
              "provider": "Mistral",
              "tier": "standard",
              "quantization": "not reported",
              "context": 262144,
              "input": 1.5,
              "output": 7.5,
              "cacheRead": null,
              "uptime1d": 99.93
            },
            {
              "model": "Mistral Medium 3.5",
              "provider": "Mistral",
              "tier": "regional",
              "quantization": "not reported",
              "context": 262144,
              "input": 1.65,
              "output": 8.25,
              "cacheRead": null,
              "uptime1d": 100
            },
            {
              "model": "Qwen3.8 Flash",
              "provider": "Alibaba",
              "tier": "standard",
              "quantization": "not reported",
              "context": 1000000,
              "input": 0.15,
              "output": 0.47,
              "cacheRead": 0.016,
              "uptime1d": 98.7
            },
            {
              "model": "Qwen3.8 Max (0902)",
              "provider": "Alibaba",
              "tier": "standard",
              "quantization": "not reported",
              "context": 1000000,
              "input": 2,
              "output": 6,
              "cacheRead": 0.25,
              "uptime1d": 99.8
            }
          ]
        }
      ],
      "related": [
        "routing-overhead",
        "cost-thought-experiments"
      ]
    },
    {
      "slug": "cost-thought-experiments",
      "title": "What if every call ran on Opus? Repricing real agent tokens",
      "seoTitle": "Agent token costs repriced: Haiku vs Sonnet vs Opus vs Fable",
      "description": "Thought experiments on real tokens: the same SWE-bench agent work priced at Haiku, Sonnet, Opus, Fable, Gemini Flash, GPT and Jev list prices.",
      "question": "Agent recorded every token it used on 33 SWE-bench instances. What would the same tokens cost at other models’ list prices, and what did caching save?",
      "answer": "Calculation, not a run. Agent used 162.9M input tokens (94.0% cache reads) and 1.8M output tokens across 33 attempts. At Sonnet 5.5 list prices that is $87.23 ($3.49 per resolved instance); the platform's own notional figure, which also counts compaction calls, is $92.64. The same tokens at Opus 5.5 prices cost $143.83, at Fable 5.1 $321.31 and at Haiku 4.5 $43.61. Without prompt caching the Sonnet bill would be $343.33. A different model would have used different tokens and resolved a different set, so these figures bound price sensitivity; they do not predict outcomes.",
      "date": "2026-10-05",
      "updated": "2026-10-05",
      "tags": [
        "thought-experiment",
        "llm-pricing",
        "prompt-caching",
        "opus",
        "sonnet",
        "haiku",
        "jev"
      ],
      "method": [
        "Tokens: the sum over every model call in the run telemetry of all 33 SWE-bench attempts (input, cache reads, one-hour cache writes, output).",
        "Prices: list prices recorded in the product price table, effective 2026-09-21 (Jev 2026-09-23, OpenAI rows 2026-10-03).",
        "Formula: uncached input × input price + cache reads × cache-read price + cache writes × write price + output × output price. Anthropic one-hour writes cost twice the input price; other vendors’ writes are priced as plain input.",
        "Cost per resolved keeps every attempt’s cost in the numerator and divides by the 25 instances Agent resolved."
      ],
      "caveats": [
        "Every repriced figure is a calculation, not a run. Only the Sonnet 5.5 row matches the model that produced the tokens.",
        "Different models use different numbers of calls, tokens and cache hits, and they resolve different instances. Use these figures for price sensitivity only.",
        "Jev is a routing model. Pricing coding tokens at Jev rates shows a floor, not a feasible configuration.",
        "The router-overhead chart uses assumed decision prompt sizes.",
        "Recorded costs are list-price estimates for subscription calls; no invoice backs them."
      ],
      "sourceIds": [
        "calc-repricing",
        "agent-swebench-c1",
        "agent-swebench-c2",
        "swebench-leaderboard",
        "price-anthropic",
        "price-google",
        "price-openai",
        "price-jev"
      ],
      "stats": [
        {
          "id": "tokens-input",
          "label": "Input tokens recorded",
          "value": 162861253,
          "unit": "tokens",
          "display": "162.9M",
          "n": 33
        },
        {
          "id": "tokens-output",
          "label": "Output tokens recorded",
          "value": 1760652,
          "unit": "tokens",
          "display": "1.8M",
          "n": 33
        },
        {
          "id": "cache-read-share",
          "label": "Share of input served from cache",
          "value": 0.9401,
          "unit": "rate",
          "display": "94.0%",
          "n": 33
        },
        {
          "id": "sonnet-repriced",
          "label": "Recorded tokens at Sonnet 5.5 list price",
          "value": 87.23,
          "unit": "usd",
          "display": "$87.23",
          "n": 33
        },
        {
          "id": "opus-repriced",
          "label": "Same tokens at Opus 5.5 list price (calculation)",
          "value": 143.83,
          "unit": "usd",
          "display": "$143.83",
          "n": 33
        },
        {
          "id": "haiku-repriced",
          "label": "Same tokens at Haiku 4.5 list price (calculation)",
          "value": 43.61,
          "unit": "usd",
          "display": "$43.61",
          "n": 33
        },
        {
          "id": "no-cache-sonnet",
          "label": "Sonnet 5.5 without caching (calculation)",
          "value": 343.33,
          "unit": "usd",
          "display": "$343.33",
          "n": 33
        },
        {
          "id": "panel-cost-per-resolved-mean",
          "label": "Public panel mean cost per resolved instance (recorded)",
          "value": 0.569,
          "unit": "usd",
          "display": "$0.57",
          "n": 11
        }
      ],
      "charts": [
        {
          "id": "repriced-cost-per-resolved",
          "title": "Thought experiment: the same tokens at other list prices",
          "subtitle": "Cost per resolved SWE-bench instance if 162.9M input and 1.8M output tokens had been billed at each model's list price",
          "kind": "bar",
          "unit": "usd",
          "yLabel": "USD per resolved instance",
          "series": [
            {
              "name": "Repriced cost per resolved instance",
              "points": [
                {
                  "label": "Claude Fable 5.1",
                  "value": 12.852,
                  "highlight": false
                },
                {
                  "label": "Claude Opus 5",
                  "value": 8.723,
                  "highlight": false
                },
                {
                  "label": "Claude Opus 5.5",
                  "value": 5.753,
                  "highlight": false
                },
                {
                  "label": "Claude Sonnet 5.5",
                  "value": 3.489,
                  "highlight": true
                },
                {
                  "label": "GPT-6.1 Sol",
                  "value": 2.097,
                  "highlight": false
                },
                {
                  "label": "Claude Haiku 4.5",
                  "value": 1.745,
                  "highlight": false
                },
                {
                  "label": "Gemini 3.x Flash",
                  "value": 1.016,
                  "highlight": false
                },
                {
                  "label": "Jev 1.13 (router)",
                  "value": 0.042,
                  "highlight": false
                }
              ]
            }
          ],
          "note": "Calculation, not a run: tokens recorded by Agent on claude-sonnet-5-5 (33 attempts, 25 resolved) times list prices effective 2026-09-21. Another model would use a different number of tokens and resolve a different set. Jev is a routing model and cannot do this work; its bar is a price floor only.",
          "sourceIds": [
            "calc-repricing",
            "agent-swebench-c1",
            "agent-swebench-c2",
            "price-anthropic",
            "price-google",
            "price-openai",
            "price-jev"
          ]
        },
        {
          "id": "cost-per-resolved-agent-vs-panel",
          "title": "Recorded cost per resolved instance: Agent vs the public panel",
          "subtitle": "Same 33 SWE-bench Verified instances; all attempts in the numerator",
          "kind": "bar",
          "unit": "usd",
          "yLabel": "USD per resolved instance",
          "series": [
            {
              "name": "Cost per resolved instance",
              "points": [
                {
                  "label": "Agent (notional)",
                  "value": 3.706,
                  "n": 25,
                  "highlight": true
                },
                {
                  "label": "Claude 4.5 Opus (high)",
                  "value": 1.184,
                  "n": 24
                },
                {
                  "label": "Claude 4.5 Sonnet (high)",
                  "value": 0.913,
                  "n": 25
                },
                {
                  "label": "Claude 4.6 Opus",
                  "value": 0.875,
                  "n": 23
                },
                {
                  "label": "GLM 5 (high)",
                  "value": 0.667,
                  "n": 26
                },
                {
                  "label": "DeepSeek V3.2 (high)",
                  "value": 0.637,
                  "n": 24
                },
                {
                  "label": "GPT 5.2 (high)",
                  "value": 0.628,
                  "n": 28
                },
                {
                  "label": "Claude 4.5 Haiku (high)",
                  "value": 0.479,
                  "n": 25
                },
                {
                  "label": "Gemini 3 Flash (high)",
                  "value": 0.436,
                  "n": 27
                },
                {
                  "label": "Kimi K2.5 (high)",
                  "value": 0.256,
                  "n": 23
                },
                {
                  "label": "MiniMax M2.5 (high)",
                  "value": 0.107,
                  "n": 23
                },
                {
                  "label": "GPT 5 mini",
                  "value": 0.08,
                  "n": 21
                }
              ]
            }
          ],
          "note": "Recorded figures, not repricing. Panel costs are published API costs for a bash-only agent. Agent's figure is a list-price estimate of subscription calls and includes onboarding, planning, verification and review.",
          "sourceIds": [
            "agent-swebench-c1",
            "agent-swebench-c2",
            "swebench-leaderboard"
          ]
        },
        {
          "id": "prompt-cache-savings",
          "title": "Thought experiment: what prompt caching saved",
          "subtitle": "The same recorded tokens with and without cache pricing",
          "kind": "grouped-bar",
          "unit": "usd",
          "yLabel": "USD for all attempts",
          "series": [
            {
              "name": "With caching (as recorded)",
              "points": [
                {
                  "label": "Claude Haiku 4.5",
                  "value": 43.61,
                  "highlight": false
                },
                {
                  "label": "Claude Sonnet 5.5",
                  "value": 87.23,
                  "highlight": true
                },
                {
                  "label": "Claude Opus 5.5",
                  "value": 143.83,
                  "highlight": false
                },
                {
                  "label": "Claude Fable 5.1",
                  "value": 321.31,
                  "highlight": false
                }
              ]
            },
            {
              "name": "Without caching",
              "points": [
                {
                  "label": "Claude Haiku 4.5",
                  "value": 171.66
                },
                {
                  "label": "Claude Sonnet 5.5",
                  "value": 343.33
                },
                {
                  "label": "Claude Opus 5.5",
                  "value": 686.66
                },
                {
                  "label": "Claude Fable 5.1",
                  "value": 1716.65
                }
              ]
            }
          ],
          "note": "94.0% of recorded input tokens were cache reads. Calculation, not a run.",
          "sourceIds": [
            "calc-repricing",
            "agent-swebench-c1",
            "agent-swebench-c2",
            "price-anthropic"
          ]
        },
        {
          "id": "token-cost-mix",
          "title": "Where the token dollars go",
          "subtitle": "Recorded tokens at Sonnet 5.5 list price, by token kind",
          "kind": "bar",
          "unit": "usd",
          "yLabel": "USD for all attempts",
          "series": [
            {
              "name": "Cost",
              "points": [
                {
                  "label": "Cache writes (1 h)",
                  "value": 38.99
                },
                {
                  "label": "Cache reads",
                  "value": 30.62
                },
                {
                  "label": "Output",
                  "value": 17.61
                },
                {
                  "label": "Uncached input",
                  "value": 0.01
                }
              ]
            }
          ],
          "note": "List-price calculation on recorded tokens: 153.1M cache reads, 9.7M cache writes, 1.8M output, 3.2k uncached input. An agent loop re-reads its context on every call, so cache reads dominate the token count; by price the largest part is cache writes (1 h).",
          "sourceIds": [
            "calc-repricing",
            "agent-swebench-c1",
            "agent-swebench-c2",
            "price-anthropic"
          ]
        },
        {
          "id": "router-overhead-jev",
          "title": "Thought experiment: the price of a routing decision on every call",
          "subtitle": "1,632 model calls; one Jev decision per call at an assumed prompt size",
          "kind": "bar",
          "unit": "usd",
          "yLabel": "USD for all attempts",
          "series": [
            {
              "name": "Jev decision cost",
              "points": [
                {
                  "label": "1k-token decision prompt",
                  "value": 0.0685
                },
                {
                  "label": "2k-token decision prompt",
                  "value": 0.1371
                },
                {
                  "label": "5k-token decision prompt",
                  "value": 0.3427
                }
              ]
            }
          ],
          "note": "Assumption-based calculation: the decision prompt sizes are assumptions, not measurements. For scale, the recorded work cost $92.64 (notional). A router only pays off if its choices save more than this.",
          "sourceIds": [
            "calc-repricing",
            "price-jev",
            "agent-swebench-c1",
            "agent-swebench-c2"
          ]
        }
      ],
      "tables": [
        {
          "id": "repricing-table",
          "title": "Repricing table (calculation)",
          "columns": [
            {
              "key": "model",
              "label": "Priced as",
              "unit": "text"
            },
            {
              "key": "vendor",
              "label": "Vendor",
              "unit": "text"
            },
            {
              "key": "inputPerM",
              "label": "Input $/M",
              "unit": "usd"
            },
            {
              "key": "cacheReadPerM",
              "label": "Cache read $/M",
              "unit": "usd"
            },
            {
              "key": "outputPerM",
              "label": "Output $/M",
              "unit": "usd"
            },
            {
              "key": "total",
              "label": "All 33 attempts",
              "unit": "usd"
            },
            {
              "key": "perAttempt",
              "label": "Per attempt",
              "unit": "usd"
            },
            {
              "key": "perResolved",
              "label": "Per resolved",
              "unit": "usd"
            }
          ],
          "rows": [
            {
              "model": "Claude Haiku 4.5",
              "vendor": "Anthropic",
              "inputPerM": 1,
              "cacheReadPerM": 0.1,
              "outputPerM": 5,
              "total": 43.61,
              "perAttempt": 1.322,
              "perResolved": 1.745
            },
            {
              "model": "Claude Sonnet 5.5",
              "vendor": "Anthropic",
              "inputPerM": 2,
              "cacheReadPerM": 0.2,
              "outputPerM": 10,
              "total": 87.23,
              "perAttempt": 2.643,
              "perResolved": 3.489
            },
            {
              "model": "Claude Opus 5.5",
              "vendor": "Anthropic",
              "inputPerM": 4,
              "cacheReadPerM": 0.2,
              "outputPerM": 20,
              "total": 143.83,
              "perAttempt": 4.359,
              "perResolved": 5.753
            },
            {
              "model": "Claude Opus 5",
              "vendor": "Anthropic",
              "inputPerM": 5,
              "cacheReadPerM": 0.5,
              "outputPerM": 25,
              "total": 218.07,
              "perAttempt": 6.608,
              "perResolved": 8.723
            },
            {
              "model": "Claude Fable 5.1",
              "vendor": "Anthropic",
              "inputPerM": 10,
              "cacheReadPerM": 0.25,
              "outputPerM": 50,
              "total": 321.31,
              "perAttempt": 9.737,
              "perResolved": 12.852
            },
            {
              "model": "Gemini 3.x Flash",
              "vendor": "Google",
              "inputPerM": 0.75,
              "cacheReadPerM": 0.075,
              "outputPerM": 3.75,
              "total": 25.4,
              "perAttempt": 0.77,
              "perResolved": 1.016
            },
            {
              "model": "GPT-6.1 Sol",
              "vendor": "OpenAI",
              "inputPerM": 2,
              "cacheReadPerM": 0.1,
              "outputPerM": 10,
              "total": 52.42,
              "perAttempt": 1.589,
              "perResolved": 2.097
            },
            {
              "model": "Jev 1.13 (router)",
              "vendor": "TypeSafe",
              "inputPerM": 0.042,
              "cacheReadPerM": 0.0042,
              "outputPerM": 0,
              "total": 1.05,
              "perAttempt": 0.032,
              "perResolved": 0.042
            }
          ]
        }
      ]
    },
    {
      "slug": "cli-model-latency-tokens",
      "title": "Claude Code CLI vs Codex CLI vs the API: latency and tokens",
      "seoTitle": "Claude Code vs Codex CLI vs API: latency and token overhead",
      "description": "194 timed runs: how long Claude Code, Codex CLI and the OpenAI API take to answer and to fix code, and how many hidden tokens a CLI adds.",
      "question": "How much time and how many tokens does a coding CLI add on top of the model, and how do Claude Code and Codex compare on the same repair task?",
      "answer": "For a one-line answer, the Codex CLI took a median 3.5 times as long as the OpenAI API with the same model and effort, and it sent about 19,551 input tokens instead of 17. On a dependency-aware scheduler repair with 296 checks, all 9 runs passed: Claude Code with Sonnet 5.5 took a median 15.0 s, the OpenAI API with GPT-6.1 Sol 17.3 s and the Codex CLI with GPT-6.1 Sol 61.2 s. With 3 to 5 runs per cell these are directional measurements, not rankings.",
      "date": "2026-10-03",
      "updated": "2026-10-05",
      "tags": [
        "claude-code",
        "codex",
        "latency",
        "tokens",
        "cli-vs-api"
      ],
      "method": [
        "Receipts from the provider explorer: each run records the route (CLI or API), the model, the effort, timings, reported tokens and a deterministic validator result.",
        "First useful output is the first streamed text that belongs to the answer. Total time runs from launch to exit, including CLI start-up.",
        "Only receipts classified \"evaluated\" count. Excluded runs (context mismatch, pilot runs, unsupported controls) and diagnostics are kept in the raw file but not charted.",
        "The matched cohort ran every configuration back to back on the same host with the same prompt."
      ],
      "caveats": [
        "Small samples: 3 to 5 runs per configuration. Medians with ranges, not intervals.",
        "The scheduler comparison pairs Claude Code with Sonnet 5.5 against GPT-6.1 Sol on Codex and the API; the routes and the models differ together.",
        "The Claude CLI receipts record only the uncached remainder of the input (2 tokens), so the Claude input column is not comparable.",
        "All runs are from one host and one network on 2026-10-03. Vendor latency changes over the day.",
        "API costs in the raw file are list-price estimates; CLI runs are subscription calls with no per-call price."
      ],
      "sourceIds": [
        "agent-provider-explorer"
      ],
      "stats": [
        {
          "id": "explorer-pass-rate",
          "label": "Evaluated runs that passed their validator",
          "value": 1,
          "unit": "rate",
          "display": "100% (194/194)",
          "n": 194,
          "ci": [
            0.9806,
            1
          ],
          "note": "18 excluded and 18 diagnostic receipts are not counted."
        },
        {
          "id": "exact-reply-cli-over-api",
          "label": "Codex CLI vs OpenAI API, median total time for a one-line answer",
          "value": 3.49,
          "unit": "ratio",
          "display": "3.5x slower",
          "n": 30,
          "note": "Medians 3.9 s (CLI) vs 1.1 s (API), same models and efforts."
        },
        {
          "id": "scheduler-claude-cli-median",
          "label": "Claude Code CLI (Sonnet 5.5) median time to repair the scheduler",
          "value": 15,
          "unit": "seconds",
          "display": "15.0 s",
          "n": 3
        },
        {
          "id": "scheduler-codex-cli-median",
          "label": "Codex CLI (GPT-6.1 Sol) median time to repair the scheduler",
          "value": 61.2,
          "unit": "seconds",
          "display": "61.2 s",
          "n": 3
        },
        {
          "id": "cli-hidden-prompt",
          "label": "Median input tokens the Codex CLI sends for a one-line request",
          "value": 19551,
          "unit": "tokens",
          "display": "19,551",
          "n": 15,
          "note": "The API sends 17 tokens for the same request."
        }
      ],
      "charts": [
        {
          "id": "cli-vs-api-exact-reply-latency",
          "title": "CLI vs API: time for a one-line answer",
          "subtitle": "Matched cohort, fixed exact reply, 5 runs per configuration",
          "kind": "dot-range",
          "unit": "seconds",
          "yLabel": "Seconds",
          "series": [
            {
              "name": "Total time",
              "points": [
                {
                  "label": "OpenAI API · GPT-6 Luna · none",
                  "value": 0.97,
                  "lo": 0.65,
                  "hi": 1.5,
                  "n": 5
                },
                {
                  "label": "OpenAI API · GPT-6.1 Sol · low",
                  "value": 1.02,
                  "lo": 0.96,
                  "hi": 1.87,
                  "n": 5
                },
                {
                  "label": "OpenAI API · GPT-6.1 Sol · high",
                  "value": 1.52,
                  "lo": 1.35,
                  "hi": 2.23,
                  "n": 5
                },
                {
                  "label": "Codex CLI · GPT-6 Luna · none",
                  "value": 3.19,
                  "lo": 2.88,
                  "hi": 3.83,
                  "n": 5
                },
                {
                  "label": "Codex CLI · GPT-6.1 Sol · low",
                  "value": 4.18,
                  "lo": 3.86,
                  "hi": 4.53,
                  "n": 5
                },
                {
                  "label": "Codex CLI · GPT-6.1 Sol · high",
                  "value": 4.19,
                  "lo": 3.81,
                  "hi": 4.69,
                  "n": 5
                }
              ]
            },
            {
              "name": "First useful output",
              "points": [
                {
                  "label": "OpenAI API · GPT-6 Luna · none",
                  "value": 0.82,
                  "lo": 0.51,
                  "hi": 1.37,
                  "n": 5
                },
                {
                  "label": "OpenAI API · GPT-6.1 Sol · low",
                  "value": 0.87,
                  "lo": 0.84,
                  "hi": 1.74,
                  "n": 5
                },
                {
                  "label": "OpenAI API · GPT-6.1 Sol · high",
                  "value": 1.34,
                  "lo": 1.26,
                  "hi": 2.12,
                  "n": 5
                },
                {
                  "label": "Codex CLI · GPT-6 Luna · none",
                  "value": 2.79,
                  "lo": 2.46,
                  "hi": 3.42,
                  "n": 5
                },
                {
                  "label": "Codex CLI · GPT-6.1 Sol · low",
                  "value": 3.75,
                  "lo": 3.44,
                  "hi": 4.1,
                  "n": 5
                },
                {
                  "label": "Codex CLI · GPT-6.1 Sol · high",
                  "value": 3.79,
                  "lo": 3.37,
                  "hi": 4.3,
                  "n": 5
                }
              ]
            }
          ],
          "note": "Dot = median; whiskers = fastest and slowest run (a range, not a confidence interval). Every run in these cohorts passed its validator.",
          "sourceIds": [
            "agent-provider-explorer"
          ]
        },
        {
          "id": "cli-vs-api-small-coding-latency",
          "title": "CLI vs API: time for a small coding task",
          "subtitle": "Matched cohort, small coding task, 3 runs per configuration",
          "kind": "dot-range",
          "unit": "seconds",
          "yLabel": "Seconds",
          "series": [
            {
              "name": "Total time",
              "points": [
                {
                  "label": "OpenAI API · GPT-6 Luna · none",
                  "value": 4.01,
                  "lo": 3.83,
                  "hi": 4.35,
                  "n": 3
                },
                {
                  "label": "OpenAI API · GPT-6.1 Sol · low",
                  "value": 6,
                  "lo": 5.44,
                  "hi": 6.2,
                  "n": 3
                },
                {
                  "label": "Codex CLI · GPT-6 Luna · none",
                  "value": 9.23,
                  "lo": 8.99,
                  "hi": 11.68,
                  "n": 3
                },
                {
                  "label": "OpenAI API · GPT-6.1 Sol · high",
                  "value": 9.56,
                  "lo": 9.44,
                  "hi": 10.94,
                  "n": 3
                },
                {
                  "label": "Codex CLI · GPT-6.1 Sol · low",
                  "value": 14.15,
                  "lo": 13.02,
                  "hi": 14.41,
                  "n": 3
                },
                {
                  "label": "Codex CLI · GPT-6.1 Sol · high",
                  "value": 17.85,
                  "lo": 17.68,
                  "hi": 22.42,
                  "n": 3
                }
              ]
            },
            {
              "name": "First useful output",
              "points": [
                {
                  "label": "OpenAI API · GPT-6 Luna · none",
                  "value": 0.67,
                  "lo": 0.62,
                  "hi": 0.81,
                  "n": 3
                },
                {
                  "label": "OpenAI API · GPT-6.1 Sol · low",
                  "value": 1.05,
                  "lo": 0.97,
                  "hi": 1.4,
                  "n": 3
                },
                {
                  "label": "Codex CLI · GPT-6 Luna · none",
                  "value": 8.68,
                  "lo": 8.27,
                  "hi": 11.01,
                  "n": 3
                },
                {
                  "label": "OpenAI API · GPT-6.1 Sol · high",
                  "value": 5.31,
                  "lo": 4.99,
                  "hi": 6.42,
                  "n": 3
                },
                {
                  "label": "Codex CLI · GPT-6.1 Sol · low",
                  "value": 13.6,
                  "lo": 12.52,
                  "hi": 13.83,
                  "n": 3
                },
                {
                  "label": "Codex CLI · GPT-6.1 Sol · high",
                  "value": 17.27,
                  "lo": 17.13,
                  "hi": 21.86,
                  "n": 3
                }
              ]
            }
          ],
          "note": "Dot = median; whiskers = fastest and slowest run (a range, not a confidence interval). Every run in these cohorts passed its validator.",
          "sourceIds": [
            "agent-provider-explorer"
          ]
        },
        {
          "id": "cli-vs-api-prompt-overhead",
          "title": "Hidden prompt: input tokens for the same one-line request",
          "subtitle": "Reported input tokens, matched cohort",
          "kind": "bar",
          "unit": "tokens",
          "yLabel": "Input tokens per call",
          "series": [
            {
              "name": "Input tokens",
              "points": [
                {
                  "label": "OpenAI API · GPT-6 Luna · none",
                  "value": 17,
                  "n": 5
                },
                {
                  "label": "OpenAI API · GPT-6.1 Sol · low",
                  "value": 17,
                  "n": 5
                },
                {
                  "label": "OpenAI API · GPT-6.1 Sol · high",
                  "value": 17,
                  "n": 5
                },
                {
                  "label": "Codex CLI · GPT-6 Luna · none",
                  "value": 18859,
                  "n": 5
                },
                {
                  "label": "Codex CLI · GPT-6.1 Sol · low",
                  "value": 19551,
                  "n": 5
                },
                {
                  "label": "Codex CLI · GPT-6.1 Sol · high",
                  "value": 19555,
                  "n": 5
                }
              ]
            }
          ],
          "note": "The CLI wraps every request in its own system prompt and tool context; the bare API sends only the request. Part of the CLI input is served from cache.",
          "sourceIds": [
            "agent-provider-explorer"
          ]
        },
        {
          "id": "scheduler-repair-claude-vs-codex",
          "title": "Repairing a scheduler: Claude Code vs Codex vs API",
          "subtitle": "Same prompt, medium effort, 296 behavioral checks, 3 runs each",
          "kind": "dot-range",
          "unit": "seconds",
          "yLabel": "Seconds",
          "series": [
            {
              "name": "Total time",
              "points": [
                {
                  "label": "Claude Code CLI · Sonnet 5.5 · medium",
                  "value": 15,
                  "lo": 13.89,
                  "hi": 15.89,
                  "n": 3,
                  "highlight": true
                },
                {
                  "label": "Codex CLI · GPT-6.1 Sol · medium",
                  "value": 61.16,
                  "lo": 59.9,
                  "hi": 69.51,
                  "n": 3
                },
                {
                  "label": "OpenAI API · GPT-6.1 Sol · medium",
                  "value": 17.32,
                  "lo": 16.28,
                  "hi": 18.61,
                  "n": 3
                }
              ]
            },
            {
              "name": "First useful output",
              "points": [
                {
                  "label": "Claude Code CLI · Sonnet 5.5 · medium",
                  "value": 7.55,
                  "lo": 6.77,
                  "hi": 7.63,
                  "n": 3
                },
                {
                  "label": "Codex CLI · GPT-6.1 Sol · medium",
                  "value": 15.56,
                  "lo": 13.65,
                  "hi": 23.04,
                  "n": 3
                },
                {
                  "label": "OpenAI API · GPT-6.1 Sol · medium",
                  "value": 7.46,
                  "lo": 6.68,
                  "hi": 9.05,
                  "n": 3
                }
              ]
            }
          ],
          "note": "All 9 runs passed all 296 checks. Dot = median, whiskers = range. Different models (Sonnet 5.5 vs GPT-6.1 Sol), so this compares route + model pairs, not routes alone.",
          "sourceIds": [
            "agent-provider-explorer"
          ]
        },
        {
          "id": "scheduler-repair-output-tokens",
          "title": "Output tokens to repair the scheduler",
          "subtitle": "Median per run; reasoning tokens shown separately where reported",
          "kind": "grouped-bar",
          "unit": "tokens",
          "yLabel": "Tokens",
          "series": [
            {
              "name": "Output tokens",
              "points": [
                {
                  "label": "Claude Code CLI · Sonnet 5.5 · medium",
                  "value": 2227,
                  "n": 3
                },
                {
                  "label": "Codex CLI · GPT-6.1 Sol · medium",
                  "value": 1181,
                  "n": 3
                },
                {
                  "label": "OpenAI API · GPT-6.1 Sol · medium",
                  "value": 1313,
                  "n": 3
                }
              ]
            },
            {
              "name": "Reasoning tokens (reported)",
              "points": [
                {
                  "label": "Claude Code CLI · Sonnet 5.5 · medium",
                  "value": 0,
                  "n": 0
                },
                {
                  "label": "Codex CLI · GPT-6.1 Sol · medium",
                  "value": 156,
                  "n": 3
                },
                {
                  "label": "OpenAI API · GPT-6.1 Sol · medium",
                  "value": 267,
                  "n": 3
                }
              ]
            }
          ],
          "note": "The Claude CLI does not report reasoning tokens separately; 0 there means \"not reported\", not \"none\".",
          "sourceIds": [
            "agent-provider-explorer"
          ]
        }
      ],
      "tables": [
        {
          "id": "cli-api-configurations",
          "title": "Every evaluated configuration",
          "columns": [
            {
              "key": "task",
              "label": "Task",
              "unit": "text"
            },
            {
              "key": "config",
              "label": "Route · model · effort",
              "unit": "text"
            },
            {
              "key": "cohort",
              "label": "Cohort",
              "unit": "text"
            },
            {
              "key": "n",
              "label": "Runs",
              "unit": "count"
            },
            {
              "key": "passed",
              "label": "Passed",
              "unit": "count"
            },
            {
              "key": "firstUseful",
              "label": "Median first useful (s)",
              "unit": "seconds"
            },
            {
              "key": "total",
              "label": "Median total (s)",
              "unit": "seconds"
            },
            {
              "key": "input",
              "label": "Median input tokens",
              "unit": "tokens"
            },
            {
              "key": "output",
              "label": "Median output tokens",
              "unit": "tokens"
            }
          ],
          "rows": [
            {
              "task": "Fixed 243-token answer",
              "config": "OpenAI API · GPT-6.1 Sol · low · flex",
              "cohort": "flex-warm-balanced",
              "n": 6,
              "passed": 6,
              "firstUseful": 1.19,
              "total": 3.29,
              "input": 279,
              "output": 243
            },
            {
              "task": "Fixed 243-token answer",
              "config": "OpenAI API · GPT-6.1 Sol · low · flex",
              "cohort": "flex-initial",
              "n": 3,
              "passed": 3,
              "firstUseful": 1.63,
              "total": 3.69,
              "input": 279,
              "output": 243
            },
            {
              "task": "Fixed 243-token answer",
              "config": "OpenAI API · GPT-6.1 Sol · low",
              "cohort": "flex-warm-balanced",
              "n": 6,
              "passed": 6,
              "firstUseful": 1.11,
              "total": 3.87,
              "input": 279,
              "output": 243
            },
            {
              "task": "Fixed 243-token answer",
              "config": "OpenAI API · GPT-6.1 Sol · low",
              "cohort": "flex-initial",
              "n": 3,
              "passed": 3,
              "firstUseful": 1.29,
              "total": 4.06,
              "input": 279,
              "output": 243
            },
            {
              "task": "Fixed 243-token answer",
              "config": "OpenAI API · GPT-6.1 Sol · low",
              "cohort": "bare-api",
              "n": 3,
              "passed": 3,
              "firstUseful": 1.37,
              "total": 4.11,
              "input": 279,
              "output": 243
            },
            {
              "task": "Fixed 243-token answer",
              "config": "Codex CLI · GPT-6.1 Sol · low",
              "cohort": "followup-stream",
              "n": 12,
              "passed": 12,
              "firstUseful": 2.45,
              "total": 7.93,
              "input": 8941,
              "output": 243
            },
            {
              "task": "Fixed 243-token answer",
              "config": "Codex CLI · GPT-6.1 Sol · low",
              "cohort": "native",
              "n": 4,
              "passed": 4,
              "firstUseful": null,
              "total": 8.86,
              "input": 4945,
              "output": 243
            },
            {
              "task": "Fixed 243-token answer",
              "config": "Codex CLI · GPT-6.1 Sol · low",
              "cohort": "followup-minimal",
              "n": 6,
              "passed": 6,
              "firstUseful": 2.75,
              "total": 9.34,
              "input": 4945,
              "output": 243
            },
            {
              "task": "Fixed exact reply",
              "config": "OpenAI API · GPT-6 Luna · none",
              "cohort": "matched",
              "n": 5,
              "passed": 5,
              "firstUseful": 0.82,
              "total": 0.97,
              "input": 17,
              "output": 7
            },
            {
              "task": "Fixed exact reply",
              "config": "OpenAI API · GPT-6.1 Sol · low",
              "cohort": "matched",
              "n": 5,
              "passed": 5,
              "firstUseful": 0.87,
              "total": 1.02,
              "input": 17,
              "output": 7
            },
            {
              "task": "Fixed exact reply",
              "config": "OpenAI API · GPT-6.1 Sol · low",
              "cohort": "parity",
              "n": 5,
              "passed": 5,
              "firstUseful": 0.94,
              "total": 1.1,
              "input": 166,
              "output": 7
            },
            {
              "task": "Fixed exact reply",
              "config": "OpenAI API · GPT-6.1 Sol · high",
              "cohort": "matched",
              "n": 5,
              "passed": 5,
              "firstUseful": 1.34,
              "total": 1.52,
              "input": 17,
              "output": 17
            },
            {
              "task": "Fixed exact reply",
              "config": "OpenAI API · GPT-6.1 Sol · low",
              "cohort": "wire-matched",
              "n": 3,
              "passed": 3,
              "firstUseful": 1.58,
              "total": 1.7,
              "input": 8297,
              "output": 7
            },
            {
              "task": "Fixed exact reply",
              "config": "OpenAI API · GPT-6.1 Sol · high",
              "cohort": "api-initial",
              "n": 1,
              "passed": 1,
              "firstUseful": 1.78,
              "total": 1.9,
              "input": 17,
              "output": 17
            },
            {
              "task": "Fixed exact reply",
              "config": "Codex CLI · GPT-6.1 Sol · low",
              "cohort": "wire-matched",
              "n": 3,
              "passed": 3,
              "firstUseful": 2.53,
              "total": 2.76,
              "input": 8297,
              "output": 7
            },
            {
              "task": "Fixed exact reply",
              "config": "Codex CLI · GPT-6.1 Sol · low",
              "cohort": "parity",
              "n": 10,
              "passed": 10,
              "firstUseful": 2.78,
              "total": 3.1,
              "input": 8296,
              "output": 7
            },
            {
              "task": "Fixed exact reply",
              "config": "Codex CLI · GPT-6 Luna · none",
              "cohort": "matched",
              "n": 5,
              "passed": 5,
              "firstUseful": 2.79,
              "total": 3.19,
              "input": 18859,
              "output": 7
            },
            {
              "task": "Fixed exact reply",
              "config": "Codex CLI · GPT-6.1 Sol · low",
              "cohort": "installed",
              "n": 6,
              "passed": 6,
              "firstUseful": 2.63,
              "total": 3.19,
              "input": 10993,
              "output": 7
            },
            {
              "task": "Fixed exact reply",
              "config": "Codex CLI · GPT-6.1 Sol · low",
              "cohort": "followup-stream",
              "n": 12,
              "passed": 12,
              "firstUseful": 2.77,
              "total": 3.42,
              "input": 8678,
              "output": 7
            },
            {
              "task": "Fixed exact reply",
              "config": "Codex CLI · GPT-6.1 Sol · low",
              "cohort": "tuning",
              "n": 6,
              "passed": 6,
              "firstUseful": 3.5,
              "total": 3.91,
              "input": 10990,
              "output": 7
            },
            {
              "task": "Fixed exact reply",
              "config": "Codex CLI · GPT-6.1 Sol · low",
              "cohort": "matched",
              "n": 5,
              "passed": 5,
              "firstUseful": 3.75,
              "total": 4.18,
              "input": 19551,
              "output": 7
            },
            {
              "task": "Fixed exact reply",
              "config": "Codex CLI · GPT-6.1 Sol · high",
              "cohort": "matched",
              "n": 5,
              "passed": 5,
              "firstUseful": 3.79,
              "total": 4.19,
              "input": 19555,
              "output": 7
            },
            {
              "task": "mergeRanges coding repair",
              "config": "OpenAI API · GPT-6.1 Sol · low · flex",
              "cohort": "flex-initial",
              "n": 1,
              "passed": 1,
              "firstUseful": 3.3,
              "total": 5.57,
              "input": 206,
              "output": 289
            },
            {
              "task": "mergeRanges coding repair",
              "config": "OpenAI API · GPT-6.1 Sol · low",
              "cohort": "flex-initial",
              "n": 1,
              "passed": 1,
              "firstUseful": 3.16,
              "total": 5.98,
              "input": 206,
              "output": 300
            },
            {
              "task": "mergeRanges coding repair",
              "config": "Codex CLI · GPT-6.1 Sol · low",
              "cohort": "followup-minimal",
              "n": 5,
              "passed": 5,
              "firstUseful": 3.28,
              "total": 6.1,
              "input": 4872,
              "output": 241
            },
            {
              "task": "mergeRanges coding repair",
              "config": "Codex CLI · GPT-6.1 Sol · low",
              "cohort": "followup-stream",
              "n": 6,
              "passed": 6,
              "firstUseful": 3.27,
              "total": 9.15,
              "input": 6549,
              "output": 256
            },
            {
              "task": "PlanJobs scheduler repair",
              "config": "Claude Code CLI · Sonnet 5.5 · medium",
              "cohort": "claude-scheduler",
              "n": 3,
              "passed": 3,
              "firstUseful": 7.55,
              "total": 15,
              "input": 2,
              "output": 2227
            },
            {
              "task": "PlanJobs scheduler repair",
              "config": "OpenAI API · GPT-6.1 Sol · medium",
              "cohort": "scheduler",
              "n": 3,
              "passed": 3,
              "firstUseful": 7.46,
              "total": 17.32,
              "input": 9563,
              "output": 1313
            },
            {
              "task": "PlanJobs scheduler repair",
              "config": "Codex CLI · GPT-6.1 Sol · medium",
              "cohort": "scheduler",
              "n": 3,
              "passed": 3,
              "firstUseful": 15.56,
              "total": 61.16,
              "input": 9563,
              "output": 1181
            },
            {
              "task": "Scratch file edit probe",
              "config": "Codex CLI · GPT-6.1 Sol · low",
              "cohort": "edit-check",
              "n": 1,
              "passed": 1,
              "firstUseful": null,
              "total": 8.34,
              "input": 35281,
              "output": 323
            },
            {
              "task": "Scratch file edit probe",
              "config": "Codex CLI · GPT-6.1 Sol · low",
              "cohort": "installed-live",
              "n": 2,
              "passed": 2,
              "firstUseful": null,
              "total": 13.99,
              "input": 18144,
              "output": 348
            },
            {
              "task": "Scratch file edit probe",
              "config": "Codex CLI · GPT-6.1 Sol · low",
              "cohort": "native",
              "n": 2,
              "passed": 2,
              "firstUseful": null,
              "total": 15.65,
              "input": 18211,
              "output": 407
            },
            {
              "task": "Synthetic coding task (API prompt)",
              "config": "OpenAI API · GPT-6.1 Sol · low",
              "cohort": "parity",
              "n": 3,
              "passed": 3,
              "firstUseful": 0.82,
              "total": 3.63,
              "input": 363,
              "output": 238
            },
            {
              "task": "Synthetic coding task (API prompt)",
              "config": "OpenAI API · GPT-6 Luna · none",
              "cohort": "matched",
              "n": 3,
              "passed": 3,
              "firstUseful": 0.67,
              "total": 4.01,
              "input": 382,
              "output": 226
            },
            {
              "task": "Synthetic coding task (API prompt)",
              "config": "OpenAI API · GPT-6.1 Sol · low",
              "cohort": "wire-matched",
              "n": 3,
              "passed": 3,
              "firstUseful": 1.5,
              "total": 4.01,
              "input": 8496,
              "output": 225
            },
            {
              "task": "Synthetic coding task (API prompt)",
              "config": "OpenAI API · GPT-6.1 Sol · low",
              "cohort": "matched",
              "n": 3,
              "passed": 3,
              "firstUseful": 1.05,
              "total": 6,
              "input": 382,
              "output": 238
            },
            {
              "task": "Synthetic coding task (API prompt)",
              "config": "OpenAI API · GPT-6.1 Sol · high",
              "cohort": "matched",
              "n": 3,
              "passed": 3,
              "firstUseful": 5.31,
              "total": 9.56,
              "input": 382,
              "output": 429
            },
            {
              "task": "Synthetic coding task (CLI matched prompt)",
              "config": "Codex CLI · GPT-6 Luna · none",
              "cohort": "matched",
              "n": 3,
              "passed": 3,
              "firstUseful": 8.68,
              "total": 9.23,
              "input": 38110,
              "output": 285
            },
            {
              "task": "Synthetic coding task (CLI matched prompt)",
              "config": "Codex CLI · GPT-6 Luna · none",
              "cohort": "tuning",
              "n": 3,
              "passed": 3,
              "firstUseful": 9.27,
              "total": 9.9,
              "input": 18244,
              "output": 261
            },
            {
              "task": "Synthetic coding task (CLI matched prompt)",
              "config": "Codex CLI · GPT-6.1 Sol · low",
              "cohort": "installed",
              "n": 6,
              "passed": 6,
              "firstUseful": 9.94,
              "total": 11.01,
              "input": 22391,
              "output": 251
            },
            {
              "task": "Synthetic coding task (CLI matched prompt)",
              "config": "Codex CLI · GPT-6.1 Sol · low",
              "cohort": "tuning",
              "n": 6,
              "passed": 6,
              "firstUseful": 13.05,
              "total": 13.65,
              "input": 22398,
              "output": 260
            },
            {
              "task": "Synthetic coding task (CLI matched prompt)",
              "config": "Codex CLI · GPT-6.1 Sol · low",
              "cohort": "matched",
              "n": 3,
              "passed": 3,
              "firstUseful": 13.6,
              "total": 14.15,
              "input": 39529,
              "output": 266
            },
            {
              "task": "Synthetic coding task (CLI matched prompt)",
              "config": "Codex CLI · GPT-6.1 Sol · high",
              "cohort": "matched",
              "n": 3,
              "passed": 3,
              "firstUseful": 17.27,
              "total": 17.85,
              "input": 39527,
              "output": 405
            },
            {
              "task": "Synthetic coding task (CLI parity prompt)",
              "config": "Codex CLI · GPT-6.1 Sol · low",
              "cohort": "wire-matched",
              "n": 3,
              "passed": 3,
              "firstUseful": 3.84,
              "total": 4.05,
              "input": 8496,
              "output": 210
            },
            {
              "task": "Synthetic coding task (CLI parity prompt)",
              "config": "Codex CLI · GPT-6.1 Sol · low",
              "cohort": "parity",
              "n": 6,
              "passed": 6,
              "firstUseful": 7.14,
              "total": 7.49,
              "input": 8495,
              "output": 210
            }
          ]
        }
      ]
    },
    {
      "slug": "coding-calibration",
      "title": "Coding calibration: what broke on three real pull requests",
      "seoTitle": "AI coding agent calibration on fastify, h3 and uvicorn tasks",
      "description": "One attempt per task, four platform builds, failures kept: how an AI worker did on real fastify/session, h3 and uvicorn issues, and what broke.",
      "question": "On three real upstream issues, does the AI worker deliver a verified change, and what stops it when it does not?",
      "answer": "Across four platform builds, verified delivery went from 0 of 3 to 1 of 3; the latest build delivered fastify/session cleanly, but h3 regressed to an empty patch when a formatting failure was lost across a context fold, and uvicorn passed its full suite yet stayed unverified for two gate-classification reasons. The latest slice cost $11.06 for the three tasks (notional). The first calibration run on 2026-10-02 produced a passing patch for fastify/session but stopped at its $5 cap before delivery. One attempt per cell finds defects; it does not measure a rate.",
      "date": "2026-10-02",
      "updated": "2026-10-05",
      "tags": [
        "calibration",
        "coding-agents",
        "failures",
        "fastify",
        "h3",
        "uvicorn"
      ],
      "method": [
        "Tasks: fastify/session #348, h3js/h3 #1533 and Kludex/uvicorn #3036, each frozen at the base commit with the merged change as reference.",
        "Fixed model claude-sonnet-5-5, routing off, balanced mode, real cold onboarding, no review replay, no operator answers.",
        "Worker and gates run offline with frozen dependencies. Gates run install, lint, typecheck, build and the full suite at base, candidate and reference, plus cross-tests.",
        "\"Functional pass\" means the frozen patch passed the gates. \"Verified delivery\" also needs a completed, clean, unassisted run.",
        "One attempt per task per slice. Every stop and failure is kept; nothing is rerun or replaced."
      ],
      "caveats": [
        "One attempt per cell: these are defect-finding runs, not rates.",
        "Each slice changes the platform build, and the last two also remove the caps. No two slices are a matched comparison.",
        "Costs are list-price estimates for subscription calls, not invoices.",
        "The f0ac3a8a h3 run was verified only after its account-route label was corrected; it counts as unverified here, as declared."
      ],
      "sourceIds": [
        "agent-coding-calibration"
      ],
      "stats": [
        {
          "id": "verified-latest",
          "label": "Verified deliveries, latest build",
          "value": 1,
          "unit": "count",
          "display": "1 of 3",
          "n": 3
        },
        {
          "id": "cost-latest",
          "label": "Notional cost, latest build, all 3 tasks",
          "value": 11.06,
          "unit": "usd",
          "display": "$11.06",
          "n": 3
        },
        {
          "id": "refusals-trend",
          "label": "Guardrail refusals, first vs latest slice",
          "value": 19,
          "unit": "count",
          "display": "26 → 19",
          "n": 3
        },
        {
          "id": "first-run-cost",
          "label": "First calibration run (capped, fastify/session)",
          "value": 4.89,
          "unit": "usd",
          "display": "$4.89, stopped at cap",
          "n": 1
        }
      ],
      "charts": [
        {
          "id": "calibration-outcomes-by-slice",
          "title": "Three real tasks, four platform builds",
          "subtitle": "Tasks per slice: functional pass, verified delivery, pull request opened",
          "kind": "grouped-bar",
          "unit": "count",
          "yLabel": "Tasks (of 3)",
          "series": [
            {
              "name": "Functional pass (offline gates)",
              "points": [
                {
                  "label": "Baseline (capped)",
                  "value": 2,
                  "n": 3
                },
                {
                  "label": "Fix wave 1 (capped)",
                  "value": 2,
                  "n": 3
                },
                {
                  "label": "Uncapped, build f0ac3a8a",
                  "value": 1,
                  "n": 3
                },
                {
                  "label": "Uncapped, build 236c0d3f",
                  "value": 1,
                  "n": 3
                }
              ]
            },
            {
              "name": "Verified delivery",
              "points": [
                {
                  "label": "Baseline (capped)",
                  "value": 0,
                  "n": 3,
                  "highlight": false
                },
                {
                  "label": "Fix wave 1 (capped)",
                  "value": 0,
                  "n": 3,
                  "highlight": false
                },
                {
                  "label": "Uncapped, build f0ac3a8a",
                  "value": 0,
                  "n": 3,
                  "highlight": false
                },
                {
                  "label": "Uncapped, build 236c0d3f",
                  "value": 1,
                  "n": 3,
                  "highlight": true
                }
              ]
            },
            {
              "name": "Pull request opened",
              "points": [
                {
                  "label": "Baseline (capped)",
                  "value": 1,
                  "n": 3
                },
                {
                  "label": "Fix wave 1 (capped)",
                  "value": 2,
                  "n": 3
                },
                {
                  "label": "Uncapped, build f0ac3a8a",
                  "value": 2,
                  "n": 3
                },
                {
                  "label": "Uncapped, build 236c0d3f",
                  "value": 2,
                  "n": 3
                }
              ]
            }
          ],
          "note": "One attempt per task per slice. The two capped slices stopped at 20 minutes or $5; the uncapped slices had no ceiling. Each slice is a different platform build, so a change is not a matched improvement.",
          "sourceIds": [
            "agent-coding-calibration"
          ]
        },
        {
          "id": "calibration-cost-by-task",
          "title": "Notional model cost per task, by slice",
          "kind": "grouped-bar",
          "unit": "usd",
          "yLabel": "USD (notional)",
          "series": [
            {
              "name": "Baseline (capped)",
              "points": [
                {
                  "label": "fastify/session",
                  "value": 3.43,
                  "highlight": false
                },
                {
                  "label": "h3js/h3",
                  "value": 4,
                  "highlight": false
                },
                {
                  "label": "Kludex/uvicorn",
                  "value": 4.88,
                  "highlight": false
                }
              ]
            },
            {
              "name": "Fix wave 1 (capped)",
              "points": [
                {
                  "label": "fastify/session",
                  "value": 4.85,
                  "highlight": false
                },
                {
                  "label": "h3js/h3",
                  "value": 3.29,
                  "highlight": false
                },
                {
                  "label": "Kludex/uvicorn",
                  "value": 4.63,
                  "highlight": false
                }
              ]
            },
            {
              "name": "Uncapped, build f0ac3a8a",
              "points": [
                {
                  "label": "fastify/session",
                  "value": 2.44,
                  "highlight": false
                },
                {
                  "label": "h3js/h3",
                  "value": 3.81,
                  "highlight": false
                },
                {
                  "label": "Kludex/uvicorn",
                  "value": 4.74,
                  "highlight": false
                }
              ]
            },
            {
              "name": "Uncapped, build 236c0d3f",
              "points": [
                {
                  "label": "fastify/session",
                  "value": 3.53,
                  "highlight": true
                },
                {
                  "label": "h3js/h3",
                  "value": 2.96,
                  "highlight": true
                },
                {
                  "label": "Kludex/uvicorn",
                  "value": 4.57,
                  "highlight": true
                }
              ]
            }
          ],
          "note": "List-price estimates of subscription calls. Failed and capped attempts count.",
          "sourceIds": [
            "agent-coding-calibration"
          ]
        },
        {
          "id": "calibration-minutes-by-task",
          "title": "Wall time per task, by slice",
          "kind": "grouped-bar",
          "unit": "minutes",
          "yLabel": "Minutes",
          "series": [
            {
              "name": "Baseline (capped)",
              "points": [
                {
                  "label": "fastify/session",
                  "value": 9,
                  "highlight": false
                },
                {
                  "label": "h3js/h3",
                  "value": 11.8,
                  "highlight": false
                },
                {
                  "label": "Kludex/uvicorn",
                  "value": 17.9,
                  "highlight": false
                }
              ]
            },
            {
              "name": "Fix wave 1 (capped)",
              "points": [
                {
                  "label": "fastify/session",
                  "value": 15.5,
                  "highlight": false
                },
                {
                  "label": "h3js/h3",
                  "value": 13.5,
                  "highlight": false
                },
                {
                  "label": "Kludex/uvicorn",
                  "value": 20.3,
                  "highlight": false
                }
              ]
            },
            {
              "name": "Uncapped, build f0ac3a8a",
              "points": [
                {
                  "label": "fastify/session",
                  "value": 7.2,
                  "highlight": false
                },
                {
                  "label": "h3js/h3",
                  "value": 12.5,
                  "highlight": false
                },
                {
                  "label": "Kludex/uvicorn",
                  "value": 22.7,
                  "highlight": false
                }
              ]
            },
            {
              "name": "Uncapped, build 236c0d3f",
              "points": [
                {
                  "label": "fastify/session",
                  "value": 11.2,
                  "highlight": true
                },
                {
                  "label": "h3js/h3",
                  "value": 9.2,
                  "highlight": true
                },
                {
                  "label": "Kludex/uvicorn",
                  "value": 19.8,
                  "highlight": true
                }
              ]
            }
          ],
          "note": "Onboarding included. uvicorn needed more than the old 20 minute cap once the caps were removed.",
          "sourceIds": [
            "agent-coding-calibration"
          ]
        },
        {
          "id": "calibration-guardrail-refusals",
          "title": "Guardrail refusals per slice",
          "subtitle": "Tool calls the platform turned back (plan schema, delivery contract, outgoing text, loops)",
          "kind": "bar",
          "unit": "count",
          "yLabel": "Refusals (3 tasks)",
          "series": [
            {
              "name": "Refusals",
              "points": [
                {
                  "label": "Baseline (capped)",
                  "value": 26,
                  "highlight": false
                },
                {
                  "label": "Fix wave 1 (capped)",
                  "value": 28,
                  "highlight": false
                },
                {
                  "label": "Uncapped, build f0ac3a8a",
                  "value": 25,
                  "highlight": false
                },
                {
                  "label": "Uncapped, build 236c0d3f",
                  "value": 19,
                  "highlight": true
                }
              ]
            }
          ],
          "note": "A refusal is a guardrail working, not always a failure: some catch real problems, some were platform defects that the next build fixed.",
          "sourceIds": [
            "agent-coding-calibration"
          ]
        }
      ],
      "tables": [
        {
          "id": "calibration-every-attempt",
          "title": "Every attempt, failures included",
          "columns": [
            {
              "key": "task",
              "label": "Task",
              "unit": "text"
            },
            {
              "key": "slice",
              "label": "Slice",
              "unit": "text"
            },
            {
              "key": "functional",
              "label": "Functional gates",
              "unit": "text"
            },
            {
              "key": "verified",
              "label": "Verified delivery",
              "unit": "text"
            },
            {
              "key": "stop",
              "label": "How it ended",
              "unit": "text"
            },
            {
              "key": "minutes",
              "label": "Minutes",
              "unit": "minutes"
            },
            {
              "key": "costUsd",
              "label": "Cost (notional)",
              "unit": "usd"
            },
            {
              "key": "calls",
              "label": "Calls",
              "unit": "calls"
            },
            {
              "key": "why",
              "label": "Why not verified",
              "unit": "text"
            }
          ],
          "rows": [
            {
              "task": "fastify/session",
              "slice": "Baseline (capped)",
              "functional": "verified",
              "verified": "no",
              "stop": "Escalated to a person",
              "minutes": 9,
              "costUsd": 3.43,
              "calls": 43,
              "why": "run-not-completed"
            },
            {
              "task": "fastify/session",
              "slice": "Fix wave 1 (capped)",
              "functional": "verified",
              "verified": "no",
              "stop": "Hit the $5 cap",
              "minutes": 15.5,
              "costUsd": 4.85,
              "calls": 73,
              "why": "integrity-caveat:cost-capped, run-not-completed"
            },
            {
              "task": "fastify/session",
              "slice": "Uncapped, build f0ac3a8a",
              "functional": "failed",
              "verified": "no",
              "stop": "Escalated to a person",
              "minutes": 7.2,
              "costUsd": 2.44,
              "calls": 29,
              "why": "account-route-mismatch, run-not-completed, acceptance-failed, cross-test-command-failed:reference:default, cross-test-failed:reference:default, empty-patch"
            },
            {
              "task": "fastify/session",
              "slice": "Uncapped, build 236c0d3f",
              "functional": "verified",
              "verified": "yes",
              "stop": "Delivered (in review)",
              "minutes": 11.2,
              "costUsd": 3.53,
              "calls": 52,
              "why": ""
            },
            {
              "task": "h3js/h3",
              "slice": "Baseline (capped)",
              "functional": "verified",
              "verified": "no",
              "stop": "Escalated to a person",
              "minutes": 11.8,
              "costUsd": 4,
              "calls": 53,
              "why": "run-not-completed"
            },
            {
              "task": "h3js/h3",
              "slice": "Fix wave 1 (capped)",
              "functional": "verified",
              "verified": "no",
              "stop": "Escalated to a person",
              "minutes": 13.5,
              "costUsd": 3.29,
              "calls": 46,
              "why": "run-not-completed"
            },
            {
              "task": "h3js/h3",
              "slice": "Uncapped, build f0ac3a8a",
              "functional": "verified",
              "verified": "no",
              "stop": "Delivered (in review)",
              "minutes": 12.5,
              "costUsd": 3.81,
              "calls": 48,
              "why": "account-route-mismatch"
            },
            {
              "task": "h3js/h3",
              "slice": "Uncapped, build 236c0d3f",
              "functional": "failed",
              "verified": "no",
              "stop": "Escalated to a person",
              "minutes": 9.2,
              "costUsd": 2.96,
              "calls": 35,
              "why": "run-not-completed, acceptance-failed, cross-test-command-failed:reference:default, cross-test-failed:reference:default, empty-patch"
            },
            {
              "task": "Kludex/uvicorn",
              "slice": "Baseline (capped)",
              "functional": "unverified",
              "verified": "no",
              "stop": "Hit the $5 cap",
              "minutes": 17.9,
              "costUsd": 4.88,
              "calls": 67,
              "why": "integrity-caveat:cost-capped, run-not-completed, acceptance-missing, base-acceptance-not-run, base-passToPass-not-passed, candidate-acceptance-not-run, cross-tests-not-run:candidate:locked, cross-tests-not-run:candidate:ws17, cross-tests-not-run:reference:locked, cross-tests-not-run:reference:ws17, step-fail:install, step-skipped:build, step-skipped:lint, step-skipped:test, step-skipped:typecheck"
            },
            {
              "task": "Kludex/uvicorn",
              "slice": "Fix wave 1 (capped)",
              "functional": "unverified",
              "verified": "no",
              "stop": "Hit the 20 min cap",
              "minutes": 20.3,
              "costUsd": 4.63,
              "calls": 68,
              "why": "run-not-completed, acceptance-missing, base-acceptance-not-run, base-passToPass-not-passed, candidate-acceptance-not-run, cross-tests-not-run:candidate:locked, cross-tests-not-run:candidate:ws17, cross-tests-not-run:reference:locked, cross-tests-not-run:reference:ws17, step-fail:install, step-skipped:build, step-skipped:lint, step-skipped:test, step-skipped:typecheck"
            },
            {
              "task": "Kludex/uvicorn",
              "slice": "Uncapped, build f0ac3a8a",
              "functional": "unverified",
              "verified": "no",
              "stop": "Delivered (in review)",
              "minutes": 22.7,
              "costUsd": 4.74,
              "calls": 73,
              "why": "account-route-mismatch, acceptance-missing, base-acceptance-not-run, base-passToPass-not-passed, candidate-acceptance-not-run, cross-tests-not-run:candidate:ws17, cross-tests-not-run:reference:ws17"
            },
            {
              "task": "Kludex/uvicorn",
              "slice": "Uncapped, build 236c0d3f",
              "functional": "unverified",
              "verified": "no",
              "stop": "Delivered (in review)",
              "minutes": 19.8,
              "costUsd": 4.57,
              "calls": 62,
              "why": "base-did-not-fail, cross-test-command-failed:candidate:ws17"
            }
          ]
        }
      ]
    },
    {
      "slug": "single-call-vs-agent-loop",
      "title": "Single call vs agent loop: does letting the model run code help? Haiku 4.5, Sonnet 5.5 and GPT-6 Luna on 8 hard tasks",
      "seoTitle": "Agent loop vs single call: does tool use improve accuracy?",
      "description": "118 attempts on 8 hard tasks: one call vs an agent loop that runs code in a sandbox. Pass rate, time, tokens and cost.",
      "question": "On 8 hard tasks with strict validators, does an agent loop that may write and run code in a sandbox pass more often than one call with tools off, and what does the loop cost in time, tokens, tool calls and list price per pass?",
      "answer": "Not clearly, on this set: no model's agent loop is ahead of its single call by the 95% intervals. Haiku 4.5: single call 11/24 (28% to 65%), agent loop 13/24 (35% to 72%); the intervals overlap, so there is no clear difference. Sonnet 5.5: single call 24/24 (86% to 100%), agent loop 16/16 (81% to 100%); both at the ceiling, so the set cannot separate them. GPT-6 Luna (Codex CLI): single call 10/16 (39% to 82%), agent loop 12/14 (60% to 96%); the intervals overlap, so there is no clear difference. Tools were optional. Scored agent-loop attempts that ran at least one tool: Haiku 4.5 24 of 24, Sonnet 5.5 3 of 16 and GPT-6 Luna (Codex CLI) 2 of 14. Median total time per attempt, single call to agent loop. Haiku 4.5: 39.0 s to 56.8 s (1.5×; the ranges overlap). Sonnet 5.5: 7.7 s to 7.4 s (about the same; the ranges overlap). GPT-6 Luna (Codex CLI): 5.2 s to 9.3 s (1.8×; the ranges overlap). Median tokens per attempt (input with cache reads, plus output), single call to agent loop. Haiku 4.5: 9,038 to 78,432. Sonnet 5.5: 3,380 to 10,483. GPT-6 Luna (Codex CLI): 11,954 to 16,058. List-price cost per strict pass, single call to agent loop. This is a calculation; the calls ran on subscriptions. Haiku 4.5: $0.0672 to $0.1422. Sonnet 5.5: $0.0143 to $0.0275. GPT-6 Luna (Codex CLI): $0.0012 to $0.0010. 2 of 56 agent-loop attempts read a file outside their work folder and are left out; 0 edits landed outside it (the CLI refused 2 outside edit or read attempts before they ran).",
      "date": "2026-10-06",
      "updated": "2026-10-06",
      "tags": [
        "agent-loop",
        "tool-use",
        "single-call",
        "hard-tasks",
        "claude-haiku",
        "claude-sonnet",
        "gpt-6-luna",
        "claude-code",
        "codex-cli",
        "sandbox"
      ],
      "method": [
        "Protocol declared before the first counted call. The same 8 tasks, prompts, validators and strict grading as the hard head-to-head (/benchmarks/hard-model-head-to-head).",
        "Single call: the task prompt, tools off, one turn, 300 s timeout. Agent loop: the same prompt plus one paragraph (\"You may create and run files in the current folder to test your answer. Your final message must be only the answer, in the format stated above.\"). The final message is graded exactly like a single call.",
        "New cells: Claude Haiku 4.5 (agent loop) · Claude Code (24); Claude Sonnet 5.5 (agent loop) · Claude Code (16); GPT-6 Luna (single call) · Codex CLI (16); GPT-6 Luna (agent loop) · Codex CLI (16). Reference cells: Claude Haiku 4.5 (single call) · Claude Code (24); Claude Sonnet 5.5 (single call) · Claude Code (24), from the hard head-to-head. One call or session at a time per account; order rep-major, then task, then configuration.",
        "Agent-loop sandbox: a fresh empty work folder per attempt, outside the temp folder. Claude Code: tools Bash, Read, Edit, Write, Glob and Grep only, the Claude Code sandbox on (writes only in the work folder, no network, no unsandboxed commands), no MCP servers, no user settings, at most 30 turns, the same 16,000-token output cap per response as the single calls. Codex CLI: exec with the workspace-write sandbox (network off), approvals never. 10-minute limit per session.",
        "Sandbox probes before the matrix (not counted): Claude Code: network blocked (the command was refused before it ran), write to the parent folder refused, file in the work folder created; Codex CLI: network blocked, write to the parent folder refused, file in the work folder created.",
        "Controls before inference: 8/8 reference answers pass, 26/26 plausible wrong answers fail, and 8/8 wrapped references are flagged as format misses.",
        "Audit of every agent-loop transcript: file-edit tool calls, read tool paths and paths in shell commands. An attempt that read a file outside its work folder is contaminated: kept in the raw extract, left out of the rates. Edits outside the work folder that ran must be 0; a call the CLI refused before it ran is counted as an attempt, not an access. The reference answers were locked (no read access) while agents ran.",
        "Stop rules: stop a route at the first usage-limit or rate-limit message; errors, time-outs and turn-limit stops count as fails. No batch stopped early and nothing was trimmed or retried.",
        "Cost per strict pass: list price × reported tokens for every attempt in the cell, divided by its strict passes. A calculation."
      ],
      "caveats": [
        "The Claude single-call cells ran in another batch on 2026-10-06 (03:23 to 04:02 UTC), with the same CLI version, tasks and validators; provider load can differ by hour.",
        "Each row is a CLI + model pair. Claude Code and Codex CLI add their own system prompts and tool schemas, and Codex CLI also loads the account’s user-level instruction file. A gap between Claude and GPT-6 Luna rows is partly the CLI.",
        "The loop changes time and tokens as well as passes. A higher pass rate that costs several times the time and tokens is a trade, not a free gain.",
        "Only 2 or 3 attempts per task and configuration (n = 14, 16 and 24 per cell). Read the intervals; per-task bars are for finding failures, not for ranking.",
        "Claude Sonnet 5.5 (single call) · Claude Code and Claude Sonnet 5.5 (agent loop) · Claude Code passed every attempt: the set has a ceiling for these configurations, so it cannot show whether the loop helps a model that already passes.",
        "List-price costs are calculations; the calls used flat subscriptions.",
        "2 agent-loop attempts read a file outside the work folder and are left out of every rate (GPT-6 Luna (agent loop) · Codex CLI: DST day-length fix r1, DST day-length fix r2), so that cell has fewer attempts and no result for those task repetitions. They stay in the raw extract. The cause was most likely the account’s user-level Codex instructions."
      ],
      "sourceIds": [
        "agent-agent-loop",
        "agent-provider-h2h-hard",
        "calc-repricing",
        "price-anthropic",
        "price-openai"
      ],
      "stats": [
        {
          "id": "agent-loop-pass-haiku",
          "label": "Claude Haiku 4.5 strict pass rate, agent loop",
          "value": 0.5417,
          "unit": "rate",
          "display": "54% (13/24)",
          "n": 24,
          "ci": [
            0.3507,
            0.7211
          ],
          "note": "Single call: 11/24 (28% to 65%); the intervals overlap, so there is no clear difference."
        },
        {
          "id": "agent-loop-pass-sonnet",
          "label": "Claude Sonnet 5.5 strict pass rate, agent loop",
          "value": 1,
          "unit": "rate",
          "display": "100% (16/16)",
          "n": 16,
          "ci": [
            0.8064,
            1
          ],
          "note": "Single call: 24/24 (86% to 100%); both at the ceiling, so the set cannot separate them."
        },
        {
          "id": "agent-loop-pass-luna",
          "label": "GPT-6 Luna strict pass rate, agent loop (Codex CLI)",
          "value": 0.8571,
          "unit": "rate",
          "display": "86% (12/14)",
          "n": 14,
          "ci": [
            0.6006,
            0.9599
          ],
          "note": "Single call: 10/16 (39% to 82%); the intervals overlap, so there is no clear difference."
        },
        {
          "id": "agent-loop-haiku-gain",
          "label": "Change in strict pass rate, agent loop minus single call, Claude Haiku 4.5 (calculation)",
          "value": 0.0833,
          "unit": "rate",
          "display": "+8 points",
          "note": "11/24 to 13/24; the intervals overlap.",
          "n": 48
        },
        {
          "id": "agent-loop-used-tools",
          "label": "Scored agent-loop attempts that ran at least one tool",
          "value": 29,
          "unit": "count",
          "display": "29 of 54",
          "note": "Haiku 4.5 24 of 24; Sonnet 5.5 3 of 16; GPT-6 Luna (Codex CLI) 2 of 14",
          "n": 54
        },
        {
          "id": "agent-loop-outside-edits",
          "label": "Edits outside the work folder that ran, in agent-loop attempts",
          "value": 0,
          "unit": "count",
          "display": "0 in 56 attempts; 2 outside attempts were refused before they ran",
          "n": 56
        },
        {
          "id": "agent-loop-contaminated",
          "label": "Agent-loop attempts left out for reading outside the work folder",
          "value": 2,
          "unit": "count",
          "display": "2 of 56",
          "n": 56
        },
        {
          "id": "agent-loop-median-tools",
          "label": "Median tool calls per agent-loop attempt",
          "value": 1.5,
          "unit": "count",
          "display": "1.5",
          "note": "Haiku 4.5 3; Sonnet 5.5 0; GPT-6 Luna (Codex CLI) 0",
          "n": 54
        }
      ],
      "charts": [
        {
          "id": "agent-loop-pass-rate",
          "title": "Strict pass rate: single call vs agent loop on eight hard tasks",
          "subtitle": "Same tasks and validators. Whiskers are 95% Wilson intervals",
          "kind": "dot-range",
          "unit": "rate",
          "polarity": "higher",
          "yLabel": "Passed",
          "series": [
            {
              "name": "Strict pass",
              "points": [
                {
                  "label": "Claude Haiku 4.5 (single call) · Claude Code",
                  "value": 0.4583,
                  "lo": 0.2789,
                  "hi": 0.6493,
                  "n": 24
                },
                {
                  "label": "Claude Haiku 4.5 (agent loop) · Claude Code",
                  "value": 0.5417,
                  "lo": 0.3507,
                  "hi": 0.7211,
                  "n": 24
                },
                {
                  "label": "Claude Sonnet 5.5 (single call) · Claude Code",
                  "value": 1,
                  "lo": 0.862,
                  "hi": 1,
                  "n": 24
                },
                {
                  "label": "Claude Sonnet 5.5 (agent loop) · Claude Code",
                  "value": 1,
                  "lo": 0.8064,
                  "hi": 1,
                  "n": 16
                },
                {
                  "label": "GPT-6 Luna (single call) · Codex CLI",
                  "value": 0.625,
                  "lo": 0.3864,
                  "hi": 0.8152,
                  "n": 16
                },
                {
                  "label": "GPT-6 Luna (agent loop) · Codex CLI",
                  "value": 0.8571,
                  "lo": 0.6006,
                  "hi": 0.9599,
                  "n": 14
                }
              ]
            }
          ],
          "note": "Whiskers are 95% Wilson intervals. A format miss never counts as a pass; an attempt with no final answer (error, time-out, turn limit) counts as a fail. Contaminated attempts (a read outside the work folder) are left out. The Claude single-call cells are reference cells from the hard head-to-head (8 tasks × 3 repetitions), reused, not rerun.",
          "whisker": "ci95",
          "sourceIds": [
            "agent-agent-loop",
            "agent-provider-h2h-hard"
          ]
        },
        {
          "id": "agent-loop-by-task",
          "title": "Strict passes per task: single call vs agent loop",
          "subtitle": "Share of attempts per task that passed strictly; 2 or 3 attempts per task and configuration",
          "kind": "grouped-bar",
          "unit": "rate",
          "polarity": "higher",
          "yLabel": "Passed",
          "series": [
            {
              "name": "Claude Haiku 4.5 (single call) · Claude Code",
              "points": [
                {
                  "label": "Interval merge fix",
                  "value": 1,
                  "lo": 0.4385,
                  "hi": 1,
                  "n": 3
                },
                {
                  "label": "DST day-length fix",
                  "value": 0.3333,
                  "lo": 0.0615,
                  "hi": 0.7923,
                  "n": 3
                },
                {
                  "label": "CSV parser",
                  "value": 0.6667,
                  "lo": 0.2077,
                  "hi": 0.9385,
                  "n": 3
                },
                {
                  "label": "Event-loop order",
                  "value": 0,
                  "lo": 0,
                  "hi": 0.5615,
                  "n": 3
                },
                {
                  "label": "Room schedule",
                  "value": 0,
                  "lo": 0,
                  "hi": 0.5615,
                  "n": 3
                },
                {
                  "label": "SemVer regex",
                  "value": 1,
                  "lo": 0.4385,
                  "hi": 1,
                  "n": 3
                },
                {
                  "label": "Money refactor",
                  "value": 0.6667,
                  "lo": 0.2077,
                  "hi": 0.9385,
                  "n": 3
                },
                {
                  "label": "SQL report",
                  "value": 0,
                  "lo": 0,
                  "hi": 0.5615,
                  "n": 3
                }
              ]
            },
            {
              "name": "Claude Haiku 4.5 (agent loop) · Claude Code",
              "points": [
                {
                  "label": "Interval merge fix",
                  "value": 0.6667,
                  "lo": 0.2077,
                  "hi": 0.9385,
                  "n": 3
                },
                {
                  "label": "DST day-length fix",
                  "value": 0.6667,
                  "lo": 0.2077,
                  "hi": 0.9385,
                  "n": 3
                },
                {
                  "label": "CSV parser",
                  "value": 0.6667,
                  "lo": 0.2077,
                  "hi": 0.9385,
                  "n": 3
                },
                {
                  "label": "Event-loop order",
                  "value": 1,
                  "lo": 0.4385,
                  "hi": 1,
                  "n": 3
                },
                {
                  "label": "Room schedule",
                  "value": 0.6667,
                  "lo": 0.2077,
                  "hi": 0.9385,
                  "n": 3
                },
                {
                  "label": "SemVer regex",
                  "value": 0.6667,
                  "lo": 0.2077,
                  "hi": 0.9385,
                  "n": 3
                },
                {
                  "label": "Money refactor",
                  "value": 0,
                  "lo": 0,
                  "hi": 0.5615,
                  "n": 3
                },
                {
                  "label": "SQL report",
                  "value": 0,
                  "lo": 0,
                  "hi": 0.5615,
                  "n": 3
                }
              ]
            },
            {
              "name": "Claude Sonnet 5.5 (single call) · Claude Code",
              "points": [
                {
                  "label": "Interval merge fix",
                  "value": 1,
                  "lo": 0.4385,
                  "hi": 1,
                  "n": 3
                },
                {
                  "label": "DST day-length fix",
                  "value": 1,
                  "lo": 0.4385,
                  "hi": 1,
                  "n": 3
                },
                {
                  "label": "CSV parser",
                  "value": 1,
                  "lo": 0.4385,
                  "hi": 1,
                  "n": 3
                },
                {
                  "label": "Event-loop order",
                  "value": 1,
                  "lo": 0.4385,
                  "hi": 1,
                  "n": 3
                },
                {
                  "label": "Room schedule",
                  "value": 1,
                  "lo": 0.4385,
                  "hi": 1,
                  "n": 3
                },
                {
                  "label": "SemVer regex",
                  "value": 1,
                  "lo": 0.4385,
                  "hi": 1,
                  "n": 3
                },
                {
                  "label": "Money refactor",
                  "value": 1,
                  "lo": 0.4385,
                  "hi": 1,
                  "n": 3
                },
                {
                  "label": "SQL report",
                  "value": 1,
                  "lo": 0.4385,
                  "hi": 1,
                  "n": 3
                }
              ]
            },
            {
              "name": "Claude Sonnet 5.5 (agent loop) · Claude Code",
              "points": [
                {
                  "label": "Interval merge fix",
                  "value": 1,
                  "lo": 0.3424,
                  "hi": 1,
                  "n": 2
                },
                {
                  "label": "DST day-length fix",
                  "value": 1,
                  "lo": 0.3424,
                  "hi": 1,
                  "n": 2
                },
                {
                  "label": "CSV parser",
                  "value": 1,
                  "lo": 0.3424,
                  "hi": 1,
                  "n": 2
                },
                {
                  "label": "Event-loop order",
                  "value": 1,
                  "lo": 0.3424,
                  "hi": 1,
                  "n": 2
                },
                {
                  "label": "Room schedule",
                  "value": 1,
                  "lo": 0.3424,
                  "hi": 1,
                  "n": 2
                },
                {
                  "label": "SemVer regex",
                  "value": 1,
                  "lo": 0.3424,
                  "hi": 1,
                  "n": 2
                },
                {
                  "label": "Money refactor",
                  "value": 1,
                  "lo": 0.3424,
                  "hi": 1,
                  "n": 2
                },
                {
                  "label": "SQL report",
                  "value": 1,
                  "lo": 0.3424,
                  "hi": 1,
                  "n": 2
                }
              ]
            },
            {
              "name": "GPT-6 Luna (single call) · Codex CLI",
              "points": [
                {
                  "label": "Interval merge fix",
                  "value": 1,
                  "lo": 0.3424,
                  "hi": 1,
                  "n": 2
                },
                {
                  "label": "DST day-length fix",
                  "value": 1,
                  "lo": 0.3424,
                  "hi": 1,
                  "n": 2
                },
                {
                  "label": "CSV parser",
                  "value": 1,
                  "lo": 0.3424,
                  "hi": 1,
                  "n": 2
                },
                {
                  "label": "Event-loop order",
                  "value": 0,
                  "lo": 0,
                  "hi": 0.6576,
                  "n": 2
                },
                {
                  "label": "Room schedule",
                  "value": 0.5,
                  "lo": 0.0945,
                  "hi": 0.9055,
                  "n": 2
                },
                {
                  "label": "SemVer regex",
                  "value": 0.5,
                  "lo": 0.0945,
                  "hi": 0.9055,
                  "n": 2
                },
                {
                  "label": "Money refactor",
                  "value": 0,
                  "lo": 0,
                  "hi": 0.6576,
                  "n": 2
                },
                {
                  "label": "SQL report",
                  "value": 1,
                  "lo": 0.3424,
                  "hi": 1,
                  "n": 2
                }
              ]
            },
            {
              "name": "GPT-6 Luna (agent loop) · Codex CLI",
              "points": [
                {
                  "label": "Interval merge fix",
                  "value": 1,
                  "lo": 0.3424,
                  "hi": 1,
                  "n": 2
                },
                {
                  "label": "CSV parser",
                  "value": 0.5,
                  "lo": 0.0945,
                  "hi": 0.9055,
                  "n": 2
                },
                {
                  "label": "Event-loop order",
                  "value": 1,
                  "lo": 0.3424,
                  "hi": 1,
                  "n": 2
                },
                {
                  "label": "Room schedule",
                  "value": 1,
                  "lo": 0.3424,
                  "hi": 1,
                  "n": 2
                },
                {
                  "label": "SemVer regex",
                  "value": 1,
                  "lo": 0.3424,
                  "hi": 1,
                  "n": 2
                },
                {
                  "label": "Money refactor",
                  "value": 0.5,
                  "lo": 0.0945,
                  "hi": 0.9055,
                  "n": 2
                },
                {
                  "label": "SQL report",
                  "value": 1,
                  "lo": 0.3424,
                  "hi": 1,
                  "n": 2
                }
              ]
            }
          ],
          "note": "Whiskers are 95% Wilson intervals on 2 or 3 attempts, so they are very wide: read this chart for where a configuration failed, not for a ranking. Contaminated attempts (a read outside the work folder) are left out.",
          "whisker": "ci95",
          "sourceIds": [
            "agent-agent-loop",
            "agent-provider-h2h-hard"
          ]
        },
        {
          "id": "agent-loop-total-time",
          "title": "Total time per attempt: single call vs agent loop",
          "subtitle": "Median per configuration; whiskers = fastest and slowest attempt",
          "kind": "dot-range",
          "unit": "seconds",
          "yLabel": "Seconds",
          "series": [
            {
              "name": "Total time per attempt",
              "points": [
                {
                  "label": "Claude Haiku 4.5 (single call) · Claude Code",
                  "value": 39.01,
                  "lo": 15.27,
                  "hi": 75.13,
                  "n": 24
                },
                {
                  "label": "Claude Haiku 4.5 (agent loop) · Claude Code",
                  "value": 56.77,
                  "lo": 24.53,
                  "hi": 223.7,
                  "n": 24
                },
                {
                  "label": "Claude Sonnet 5.5 (single call) · Claude Code",
                  "value": 7.75,
                  "lo": 2.26,
                  "hi": 34.79,
                  "n": 24
                },
                {
                  "label": "Claude Sonnet 5.5 (agent loop) · Claude Code",
                  "value": 7.41,
                  "lo": 2.75,
                  "hi": 24.19,
                  "n": 16
                },
                {
                  "label": "GPT-6 Luna (single call) · Codex CLI",
                  "value": 5.16,
                  "lo": 3.59,
                  "hi": 11.32,
                  "n": 16
                },
                {
                  "label": "GPT-6 Luna (agent loop) · Codex CLI",
                  "value": 9.32,
                  "lo": 3.78,
                  "hi": 15.89,
                  "n": 14
                }
              ]
            }
          ],
          "note": "Whiskers are a range (fastest and slowest attempt), not a confidence interval. Agent loop: wall time of the whole session, failures and time-outs included. One host, one network. The Claude single-call cells are reference cells from the hard head-to-head (8 tasks × 3 repetitions), reused, not rerun. The reference cells ran in a different hour.",
          "whisker": "minmax",
          "sourceIds": [
            "agent-agent-loop",
            "agent-provider-h2h-hard"
          ]
        },
        {
          "id": "agent-loop-tokens",
          "title": "Tokens per attempt: single call vs agent loop",
          "subtitle": "Median per configuration; whiskers = fewest and most",
          "kind": "grouped-bar",
          "unit": "tokens",
          "yLabel": "Tokens",
          "series": [
            {
              "name": "Input tokens (cache reads included)",
              "points": [
                {
                  "label": "Claude Haiku 4.5 (single call) · Claude Code",
                  "value": 3941,
                  "lo": 3879,
                  "hi": 4221,
                  "n": 24
                },
                {
                  "label": "Claude Haiku 4.5 (agent loop) · Claude Code",
                  "value": 71691,
                  "lo": 41732,
                  "hi": 516306,
                  "n": 24
                },
                {
                  "label": "Claude Sonnet 5.5 (single call) · Claude Code",
                  "value": 2281,
                  "lo": 2234,
                  "hi": 2669,
                  "n": 24
                },
                {
                  "label": "Claude Sonnet 5.5 (agent loop) · Claude Code",
                  "value": 9550,
                  "lo": 9398,
                  "hi": 33040,
                  "n": 16
                },
                {
                  "label": "GPT-6 Luna (single call) · Codex CLI",
                  "value": 11582,
                  "lo": 11526,
                  "hi": 11818,
                  "n": 16
                },
                {
                  "label": "GPT-6 Luna (agent loop) · Codex CLI",
                  "value": 15530,
                  "lo": 15391,
                  "hi": 39009,
                  "n": 14
                }
              ]
            },
            {
              "name": "Output tokens",
              "points": [
                {
                  "label": "Claude Haiku 4.5 (single call) · Claude Code",
                  "value": 5064,
                  "lo": 1899,
                  "hi": 9321,
                  "n": 24
                },
                {
                  "label": "Claude Haiku 4.5 (agent loop) · Claude Code",
                  "value": 7912,
                  "lo": 2541,
                  "hi": 20654,
                  "n": 24
                },
                {
                  "label": "Claude Sonnet 5.5 (single call) · Claude Code",
                  "value": 1050,
                  "lo": 176,
                  "hi": 3895,
                  "n": 24
                },
                {
                  "label": "Claude Sonnet 5.5 (agent loop) · Claude Code",
                  "value": 876,
                  "lo": 219,
                  "hi": 3243,
                  "n": 16
                },
                {
                  "label": "GPT-6 Luna (single call) · Codex CLI",
                  "value": 345,
                  "lo": 36,
                  "hi": 634,
                  "n": 16
                },
                {
                  "label": "GPT-6 Luna (agent loop) · Codex CLI",
                  "value": 480,
                  "lo": 143,
                  "hi": 858,
                  "n": 14
                }
              ]
            }
          ],
          "note": "Whiskers are a range (fewest and most), not a confidence interval. Input counts the whole prompt of every model request in the attempt, cache reads and writes included; an agent loop re-sends its growing context each turn. Codex input includes its own system prompt and tool schemas. More tokens is not better or worse by itself.",
          "whisker": "minmax",
          "sourceIds": [
            "agent-agent-loop",
            "agent-provider-h2h-hard"
          ]
        },
        {
          "id": "agent-loop-tool-calls",
          "title": "Tool calls per agent-loop attempt",
          "subtitle": "Median per configuration; whiskers = fewest and most. A single call makes none",
          "kind": "dot-range",
          "unit": "count",
          "yLabel": "Tool calls",
          "series": [
            {
              "name": "Tool calls per attempt",
              "points": [
                {
                  "label": "Claude Haiku 4.5 (agent loop) · Claude Code",
                  "value": 3,
                  "lo": 2,
                  "hi": 18,
                  "n": 24
                },
                {
                  "label": "Claude Sonnet 5.5 (agent loop) · Claude Code",
                  "value": 0,
                  "lo": 0,
                  "hi": 3,
                  "n": 16
                },
                {
                  "label": "GPT-6 Luna (agent loop) · Codex CLI",
                  "value": 0,
                  "lo": 0,
                  "hi": 1,
                  "n": 14
                }
              ]
            }
          ],
          "note": "Whiskers are a range (fewest and most), not a confidence interval. Claude Code tools: shell, read, edit, write, glob, grep. Codex CLI: shell commands and file changes. The model chose whether to test its answer; the prompt allowed it but did not require it.",
          "whisker": "minmax",
          "sourceIds": [
            "agent-agent-loop"
          ]
        },
        {
          "id": "agent-loop-cost-per-pass",
          "title": "List-price cost per strict pass: single call vs agent loop (calculation)",
          "subtitle": "All attempts in a configuration divided by its strict passes",
          "kind": "bar",
          "unit": "usd",
          "yLabel": "USD per strict pass",
          "series": [
            {
              "name": "Cost per strict pass",
              "points": [
                {
                  "label": "Claude Haiku 4.5 (single call) · Claude Code",
                  "value": 0.0672,
                  "n": 24
                },
                {
                  "label": "Claude Haiku 4.5 (agent loop) · Claude Code",
                  "value": 0.14225,
                  "n": 24
                },
                {
                  "label": "Claude Sonnet 5.5 (single call) · Claude Code",
                  "value": 0.01435,
                  "n": 24
                },
                {
                  "label": "Claude Sonnet 5.5 (agent loop) · Claude Code",
                  "value": 0.02746,
                  "n": 16
                },
                {
                  "label": "GPT-6 Luna (single call) · Codex CLI",
                  "value": 0.00116,
                  "n": 16
                },
                {
                  "label": "GPT-6 Luna (agent loop) · Codex CLI",
                  "value": 0.00099,
                  "n": 14
                }
              ]
            }
          ],
          "note": "Calculation, not a bill: reported tokens × list price for every attempt (per model when a session used several), cache reads at the cache-read price and cache writes at the one-hour write price, divided by the strict passes. The calls ran on flat subscriptions.",
          "sourceIds": [
            "agent-agent-loop",
            "agent-provider-h2h-hard",
            "calc-repricing",
            "price-anthropic",
            "price-openai"
          ]
        }
      ],
      "tables": [
        {
          "id": "agent-loop-cells",
          "title": "Every single-call and agent-loop cell",
          "columns": [
            {
              "key": "config",
              "label": "Configuration",
              "unit": "text"
            },
            {
              "key": "origin",
              "label": "Cell",
              "unit": "text"
            },
            {
              "key": "strict",
              "label": "Strict passes",
              "unit": "text"
            },
            {
              "key": "ci",
              "label": "95% interval",
              "unit": "text"
            },
            {
              "key": "lenient",
              "label": "Answer correct (lenient)",
              "unit": "text"
            },
            {
              "key": "formatMisses",
              "label": "Format misses",
              "unit": "count"
            },
            {
              "key": "wrong",
              "label": "Wrong answers",
              "unit": "count"
            },
            {
              "key": "noAnswer",
              "label": "No answer (error, time-out, turn limit)",
              "unit": "count"
            },
            {
              "key": "medianTotal",
              "label": "Median total (s)",
              "unit": "seconds"
            },
            {
              "key": "rangeTotal",
              "label": "Fastest to slowest (s)",
              "unit": "text"
            },
            {
              "key": "medianTools",
              "label": "Median tool calls",
              "unit": "count"
            },
            {
              "key": "usedTools",
              "label": "Attempts that ran a tool",
              "unit": "text"
            },
            {
              "key": "medianTokens",
              "label": "Median total tokens",
              "unit": "tokens"
            },
            {
              "key": "perPass",
              "label": "USD per strict pass (calculation)",
              "unit": "usd"
            },
            {
              "key": "contaminated",
              "label": "Contaminated (left out)",
              "unit": "count"
            }
          ],
          "rows": [
            {
              "config": "Claude Haiku 4.5 (single call) · Claude Code",
              "origin": "reference (hard head-to-head)",
              "strict": "11/24",
              "ci": "28% to 65%",
              "lenient": "16/24",
              "formatMisses": 5,
              "wrong": 8,
              "noAnswer": 0,
              "medianTotal": 39.01,
              "rangeTotal": "15.3 to 75.1",
              "medianTools": 0,
              "usedTools": "tools off",
              "medianTokens": 9038,
              "perPass": 0.0672,
              "contaminated": 0
            },
            {
              "config": "Claude Haiku 4.5 (agent loop) · Claude Code",
              "origin": "new run",
              "strict": "13/24",
              "ci": "35% to 72%",
              "lenient": "20/24",
              "formatMisses": 7,
              "wrong": 4,
              "noAnswer": 0,
              "medianTotal": 56.77,
              "rangeTotal": "24.5 to 223.7",
              "medianTools": 3,
              "usedTools": "24/24",
              "medianTokens": 78432,
              "perPass": 0.14225,
              "contaminated": 0
            },
            {
              "config": "Claude Sonnet 5.5 (single call) · Claude Code",
              "origin": "reference (hard head-to-head)",
              "strict": "24/24",
              "ci": "86% to 100%",
              "lenient": "24/24",
              "formatMisses": 0,
              "wrong": 0,
              "noAnswer": 0,
              "medianTotal": 7.75,
              "rangeTotal": "2.3 to 34.8",
              "medianTools": 0,
              "usedTools": "tools off",
              "medianTokens": 3380,
              "perPass": 0.01435,
              "contaminated": 0
            },
            {
              "config": "Claude Sonnet 5.5 (agent loop) · Claude Code",
              "origin": "new run",
              "strict": "16/16",
              "ci": "81% to 100%",
              "lenient": "16/16",
              "formatMisses": 0,
              "wrong": 0,
              "noAnswer": 0,
              "medianTotal": 7.41,
              "rangeTotal": "2.7 to 24.2",
              "medianTools": 0,
              "usedTools": "3/16",
              "medianTokens": 10483,
              "perPass": 0.02746,
              "contaminated": 0
            },
            {
              "config": "GPT-6 Luna (single call) · Codex CLI",
              "origin": "new run",
              "strict": "10/16",
              "ci": "39% to 82%",
              "lenient": "11/16",
              "formatMisses": 1,
              "wrong": 5,
              "noAnswer": 0,
              "medianTotal": 5.16,
              "rangeTotal": "3.6 to 11.3",
              "medianTools": 0,
              "usedTools": "tools off",
              "medianTokens": 11954,
              "perPass": 0.00116,
              "contaminated": 0
            },
            {
              "config": "GPT-6 Luna (agent loop) · Codex CLI",
              "origin": "new run",
              "strict": "12/14",
              "ci": "60% to 96%",
              "lenient": "12/14",
              "formatMisses": 0,
              "wrong": 2,
              "noAnswer": 0,
              "medianTotal": 9.32,
              "rangeTotal": "3.8 to 15.9",
              "medianTools": 0,
              "usedTools": "2/14",
              "medianTokens": 16058,
              "perPass": 0.00099,
              "contaminated": 2
            }
          ]
        },
        {
          "id": "agent-loop-audit",
          "title": "Sandbox audit of every agent-loop attempt",
          "columns": [
            {
              "key": "config",
              "label": "Configuration",
              "unit": "text"
            },
            {
              "key": "attempts",
              "label": "Attempts",
              "unit": "count"
            },
            {
              "key": "outsideEdits",
              "label": "Edits outside the work folder that ran",
              "unit": "count"
            },
            {
              "key": "refused",
              "label": "Outside edit or read attempts refused before they ran",
              "unit": "count"
            },
            {
              "key": "contaminated",
              "label": "Contaminated attempts (read outside the work folder)",
              "unit": "count"
            },
            {
              "key": "which",
              "label": "Which (task, repetition)",
              "unit": "text"
            },
            {
              "key": "timedOut",
              "label": "Time-outs (10 min)",
              "unit": "count"
            },
            {
              "key": "maxTurns",
              "label": "Stopped at 30 turns",
              "unit": "count"
            }
          ],
          "rows": [
            {
              "config": "Claude Haiku 4.5 (agent loop) · Claude Code",
              "attempts": 24,
              "outsideEdits": 0,
              "refused": 2,
              "contaminated": 0,
              "which": "none",
              "timedOut": 0,
              "maxTurns": 0
            },
            {
              "config": "Claude Sonnet 5.5 (agent loop) · Claude Code",
              "attempts": 16,
              "outsideEdits": 0,
              "refused": 0,
              "contaminated": 0,
              "which": "none",
              "timedOut": 0,
              "maxTurns": 0
            },
            {
              "config": "GPT-6 Luna (agent loop) · Codex CLI",
              "attempts": 16,
              "outsideEdits": 0,
              "refused": 0,
              "contaminated": 2,
              "which": "DST day-length fix r1, DST day-length fix r2",
              "timedOut": 0,
              "maxTurns": 0
            }
          ]
        }
      ],
      "related": [
        "hard-model-head-to-head",
        "cli-model-latency-tokens"
      ]
    },
    {
      "slug": "haiku-thinking-on-off",
      "title": "Does thinking pay for Claude Haiku 4.5? Thinking on vs off",
      "seoTitle": "Claude Haiku 4.5 thinking on vs off: benchmark results",
      "description": "Claude Haiku 4.5 with extended thinking on and off: 82 routing decisions and 8 hard tasks. Accuracy with 95% intervals, time and cost.",
      "question": "Does extended thinking pay for Claude Haiku 4.5: what does it buy in accuracy, and what does it cost in time and money, on typed routing decisions and on hard tasks?",
      "answer": "Routing accuracy is unresolved on this sample of 82 paired decisions. Thinking off answered 87% (71/82; 95% interval 78% to 92%) exactly. Thinking on answered 89% (73/82; 95% interval 80% to 94%) exactly. Only thinking on was right in 6 cases; only thinking off in 4. Both were right in 67 cases and wrong in 5. The exact McNemar test gives p = 0.754. The test does not show a difference. This does not establish equal accuracy. The two 95% intervals overlap. Observed median wall time: 4.66 s with thinking off and 12.54 s with thinking on. The thinking-on median is 2.7 times the thinking-off median (a calculation). Wall-time ranges: 2.20 s to 11.57 s off; 5.86 s to 51.28 s on. These ranges overlap and are not confidence intervals. The 95th percentiles are 8.18 s off and 34.48 s on. Mean thinking tokens per decision: 0 off and 1,101 on. List-price cost per 1,000 decisions: $3.36 off and $8.92 on (a calculation). The recorded Sonnet 5.5 low-effort reference answered 94% (77/82; 95% interval 87% to 97%) exactly. Its median wall time was 2.60 s, with range 1.99 s to 5.58 s (not an interval). Only Sonnet was right in 7 cases; only thinking-off Haiku in 1. Exact McNemar p = 0.07 (secondary test). Hard tasks: 8 tasks, each repeated 3 times per arm. Thinking off passed 17% (4/24; 95% interval 7% to 36%) strictly. Thinking on passed 46% (11/24; 95% interval 28% to 65%) strictly. Format misses: 0 off and 5 on. These are right answers in the wrong wrapping, never strict passes. The strict 95% intervals overlap, so strict pass rate does not separate the arms. Lenient passes count format misses: 17% (4/24; 95% interval 7% to 36%) off; 67% (16/24; 95% interval 47% to 82%) on. The lenient intervals do not overlap: thinking on is ahead on this reading. Median total time per call: 2.9 s off and 39.0 s on. Single-call ranges: 1.7 s to 13.0 s off; 15.3 s to 75.1 s on. These ranges do not overlap and are not confidence intervals. List-price cost per strict pass: $0.0365 off and $0.0672 on (a calculation). The calculation divides all call costs by 4 and 11 strict passes. It has no interval. Thinking off was cheaper per strict pass in this sample, but it passed fewer calls. Both thinking-on arms reuse earlier recorded sessions, so the timing comparison is not concurrent.",
      "date": "2026-10-06",
      "updated": "2026-10-06",
      "tags": [
        "claude-haiku",
        "extended-thinking",
        "routing",
        "hard-tasks",
        "latency",
        "cost"
      ],
      "method": [
        "Routing uses the platform’s labelled decision suites: Failure class (failure-class-2026-10-05.1, 18 cases); Message intent (frontdoor-intent-2026-10-04.2, 20 cases); Is it a rule? (memory-is-rule-2026-10-04.2, 12 cases); Context shape (context-shape-2026-10-05.1, 32 cases).",
        "Production rules select the cases and leave the scored keys open. This gives 82 decisions and 194 scored questions.",
        "Each decision uses one Claude Code CLI call. Calls use JSON output, tools off, no MCP and no saved session. They use the production system text, prompt and JSON schema. Calls run one at a time; temperature cannot be set.",
        "The decision-eval runner scores each reply. Exact means every scored question is acceptable. Per-question accuracy counts each scored question. Errors and unanswered questions count as wrong.",
        "Thinking off sets MAX_THINKING_TOKENS=0 on the claude process. Two uncounted probes check the setting, one per route. The probes and all new counted calls reported zero thinking tokens.",
        "Thinking off uses new calls. Thinking on uses the CLI default in the recorded routing run. Sonnet 5.5 uses low effort in that recorded run. Both reference arms are reused; they were not rerun.",
        "The exact two-sided McNemar test uses case-level pairs that differ (binomial, p = 0.5). Rate intervals are 95% Wilson intervals.",
        "Timing uses wall time and the CLI’s reported API time. The same code computes nearest-rank p50 and p95 from each arm’s call log. Cost uses reported input, cache and output tokens times list price (a calculation).",
        "Hard tasks use the 8 unchanged tasks and sandboxed deterministic validators from the hard head-to-head.",
        "Controls ran before the first new model call, but before protocol creation. All 8 reference answers passed (8 tested). All 26 planted wrong answers failed. All 8 wrapped references were flagged as format misses.",
        "Strict pass means the whole reply passes. A format miss means a lenient extractor finds a passing answer; it never counts as a strict pass. Errors and incomplete attempts count as failures.",
        "Thinking off used 24 calls: 8 tasks, three repetitions each. Calls ran one at a time, with no effort flag, tools off, no MCP and one turn. The timeout was 300 s; the output-token cap was 16,000. Thinking on reuses 24 recorded default Haiku receipts.",
        "Cost per strict pass divides all call costs in a cell by its strict passes. Each call cost uses list price times reported tokens (a calculation).",
        "Recorded order: controls at 2026-10-06T21:27:56.463Z; protocol creation at 2026-10-06T21:29:01.607477Z; then the two probes. The first new counted call started at 2026-10-06T21:29:57.270Z. Reused reference calls predate both the controls and this protocol.",
        "Declared caps: 106 new counted calls and 112 calls including probes. Observed: 106 counted calls plus 2 probes; the total stayed within both call caps.",
        "Attempts: 106 counted calls, 2 uncounted probes. Nothing was trimmed or retried, and no run hit a usage limit. Every attempt is in the published extract."
      ],
      "caveats": [
        "The thinking-on and Sonnet arms are the recorded routing run of 2026-10-05, reused. The thinking-off arm ran on a different day and on a different Claude subscription account, so this is a confounded comparison. Day, account, CLI version and input-token differences can affect the results. The data cannot isolate the effect of thinking.",
        "82 decisions in four small hand-labelled sets: intervals are wide, and a paired test with few discordant cases has little power. The case sets and question wording were tuned in fix waves against Jev answers (2026-10-04 to 2026-10-05).",
        "Thinking off here means MAX_THINKING_TOKENS=0 on the Claude Code process, checked by the CLI’s own thinking-token counters. It says nothing about the API’s thinking settings or about other models.",
        "Claude Code adds its own start-up time and tool-schema tokens to every call; a direct API router would skip them. CLI timings include that overhead.",
        "List-price costs are calculations from reported tokens; the calls used a flat subscription. Sonnet cache writes use the one-hour rate ($4 per million tokens), as recorded in its receipts. The calculation matches the CLI-reported cost.",
        "The Haiku routing arms have the same prompt character count for every case, but thinking-on calls report about 297 more input tokens per call (1,831 against 1,534, a calculation from the two means). The cause is unknown, and the CLI version of the thinking-on run is not recorded. At Haiku’s list price that is about $0.30 of the $5.56 cost gap per 1,000 decisions (a calculation).",
        "Hard tasks: 24 calls per cell (3 per task); a 4/24 result has a 95% interval of 7% to 36%. The thinking-on cell is the recorded hard head-to-head receipts (batch of 2026-10-06, effort flag not passed), reused; the thinking-off cell ran later, on another account.",
        "These are eight distinct hard tasks, repeated three times each. Repeats on the same task are related. Call-level Wilson intervals are descriptive; they do not measure uncertainty across unseen tasks.",
        "Strict format rules decide part of the hard result: a reply in a code fence fails strictly. The lenient reading is shown next to it so the two can be told apart.",
        "Cost per strict pass divides a cell’s cost by its strict passes (4 and 11). It has no interval, and the strict pass-rate intervals overlap, so the order of the two costs per pass is not settled.",
        "Both arms hit a strict-pass ceiling on Fix an interval-merge function (off-by-one and edge cases): 100% (3/3; 95% interval 44% to 100%) in each arm. Repeats on these tasks cannot establish equal accuracy on harder variants.",
        "Controls and reused calls predate protocol creation. The file predates the new probes and counted calls, but later amendments changed it. No frozen initial copy verifies the original wording.",
        "The run folder has no saved pre-batch usage-gate readings. Later corroboration does not prove that each pre-batch gate ran.",
        "The runs used a shared Mac. Host load was not recorded; the run does not prove isolation. Timing gaps do not establish a cause.",
        "Routing has a ceiling on these subsets: Message intent, 100% (20/20; 95% interval 84% to 100%) in every arm; Is it a rule?, 100% (12/12; 95% interval 76% to 100%) in every arm. These results cannot establish equal accuracy on harder cases."
      ],
      "sourceIds": [
        "agent-haiku-thinking",
        "agent-routing",
        "calc-repricing",
        "price-anthropic",
        "agent-provider-h2h-hard"
      ],
      "hero": {
        "statIds": [
          "haiku-thinking-exact-off",
          "haiku-thinking-exact-on"
        ],
        "testStatId": "haiku-thinking-mcnemar"
      },
      "stats": [
        {
          "id": "haiku-thinking-exact-off",
          "label": "Claude Haiku 4.5 (thinking off): exact routing decisions",
          "value": 0.8659,
          "unit": "rate",
          "display": "87% (71/82)",
          "n": 82,
          "ci": [
            0.7755,
            0.9234
          ]
        },
        {
          "id": "haiku-thinking-exact-on",
          "label": "Claude Haiku 4.5 (thinking on): exact routing decisions",
          "value": 0.8902,
          "unit": "rate",
          "display": "89% (73/82)",
          "n": 82,
          "ci": [
            0.8044,
            0.9412
          ]
        },
        {
          "id": "haiku-thinking-keys-off",
          "label": "Claude Haiku 4.5 (thinking off): per-question routing accuracy",
          "value": 0.9124,
          "unit": "rate",
          "display": "91% (177/194)",
          "n": 194,
          "ci": [
            0.8642,
            0.9446
          ]
        },
        {
          "id": "haiku-thinking-keys-on",
          "label": "Claude Haiku 4.5 (thinking on): per-question routing accuracy",
          "value": 0.9433,
          "unit": "rate",
          "display": "94% (183/194)",
          "n": 194,
          "ci": [
            0.9013,
            0.968
          ]
        },
        {
          "id": "haiku-thinking-mcnemar",
          "label": "Paired exact test, thinking off vs on (exact McNemar p)",
          "value": 0.754,
          "unit": "score",
          "display": "0.754",
          "note": "p = 0.754 (6 only on, 4 only off, 82 cases)",
          "n": 82
        },
        {
          "id": "haiku-thinking-wall-off",
          "label": "Claude Haiku 4.5 (thinking off): median wall time per routing decision",
          "value": 4.66,
          "unit": "seconds",
          "display": "4.66 s",
          "n": 82,
          "note": "Range 2.20 s to 11.57 s; not a confidence interval. Nearest-rank p50."
        },
        {
          "id": "haiku-thinking-wall-on",
          "label": "Claude Haiku 4.5 (thinking on): median wall time per routing decision",
          "value": 12.54,
          "unit": "seconds",
          "display": "12.54 s",
          "n": 82,
          "note": "Range 5.86 s to 51.28 s; not a confidence interval. Nearest-rank p50."
        },
        {
          "id": "haiku-thinking-wall-ratio",
          "label": "Median wall time, thinking on ÷ thinking off (calculation)",
          "value": 2.69,
          "unit": "ratio",
          "display": "2.7x (12.54 s ÷ 4.66 s)",
          "n": 82,
          "note": "Calculation from two medians; the arms did not run at the same time."
        },
        {
          "id": "haiku-thinking-tokens-off",
          "label": "Claude Haiku 4.5 (thinking off): thinking tokens per routing decision (mean)",
          "value": 0,
          "unit": "tokens",
          "display": "0",
          "n": 82
        },
        {
          "id": "haiku-thinking-tokens-on",
          "label": "Claude Haiku 4.5 (thinking on): thinking tokens per routing decision (mean)",
          "value": 1100.8,
          "unit": "tokens",
          "display": "1,101 (range 285 to 4,443)",
          "n": 82
        },
        {
          "id": "haiku-thinking-cost-off",
          "label": "Claude Haiku 4.5 (thinking off): list-price cost per 1,000 routing decisions (calculation)",
          "value": 3.364,
          "unit": "usd",
          "display": "$3.364",
          "n": 82
        },
        {
          "id": "haiku-thinking-cost-on",
          "label": "Claude Haiku 4.5 (thinking on): list-price cost per 1,000 routing decisions (calculation)",
          "value": 8.924,
          "unit": "usd",
          "display": "$8.924",
          "n": 82
        },
        {
          "id": "haiku-thinking-hard-strict-off",
          "label": "Claude Haiku 4.5 (thinking off): strict passes on hard tasks",
          "value": 0.1667,
          "unit": "rate",
          "display": "17% (4/24)",
          "n": 24,
          "ci": [
            0.0668,
            0.3585
          ]
        },
        {
          "id": "haiku-thinking-hard-strict-on",
          "label": "Claude Haiku 4.5 (thinking on): strict passes on hard tasks",
          "value": 0.4583,
          "unit": "rate",
          "display": "46% (11/24)",
          "n": 24,
          "ci": [
            0.2789,
            0.6493
          ]
        },
        {
          "id": "haiku-thinking-hard-time-off",
          "label": "Claude Haiku 4.5 (thinking off): median total time per hard-task call",
          "value": 2.95,
          "unit": "seconds",
          "display": "2.9 s",
          "n": 24,
          "note": "Range 1.7 s to 13.0 s; not a confidence interval."
        },
        {
          "id": "haiku-thinking-hard-time-on",
          "label": "Claude Haiku 4.5 (thinking on): median total time per hard-task call",
          "value": 39.01,
          "unit": "seconds",
          "display": "39.0 s",
          "n": 24,
          "note": "Range 15.3 s to 75.1 s; not a confidence interval."
        },
        {
          "id": "haiku-thinking-hard-reasoning-on",
          "label": "Claude Haiku 4.5 (thinking on): median reasoning tokens per hard-task call",
          "value": 4556,
          "unit": "tokens",
          "display": "4,556",
          "n": 24
        },
        {
          "id": "haiku-thinking-hard-cost-per-pass-off",
          "label": "Claude Haiku 4.5 (thinking off): list-price cost per strict pass on hard tasks (calculation)",
          "value": 0.03654,
          "unit": "usd",
          "display": "$0.0365",
          "n": 24
        },
        {
          "id": "haiku-thinking-hard-cost-per-pass-on",
          "label": "Claude Haiku 4.5 (thinking on): list-price cost per strict pass on hard tasks (calculation)",
          "value": 0.0672,
          "unit": "usd",
          "display": "$0.0672",
          "n": 24
        },
        {
          "id": "haiku-thinking-new-calls",
          "label": "Counted model calls made for this study (thinking off)",
          "value": 106,
          "unit": "calls",
          "display": "106 (82 routing, 24 hard tasks; 2 more uncounted probes)",
          "n": 106
        },
        {
          "id": "haiku-thinking-off-check",
          "label": "Thinking-off calls that reported any thinking tokens",
          "value": 0,
          "unit": "count",
          "display": "0 of 106",
          "n": 106,
          "note": "Each call reports its thinking tokens in the CLI result; 0 means the setting held."
        }
      ],
      "charts": [
        {
          "id": "haiku-thinking-router-exact",
          "title": "Haiku thinking study: typed routing decisions answered exactly right",
          "subtitle": "Claude Haiku 4.5 and Claude Sonnet 5.5 (low effort); 82 decisions, the same cases for every arm",
          "kind": "dot-range",
          "unit": "rate",
          "polarity": "higher",
          "yLabel": "Correct",
          "series": [
            {
              "name": "Exact decisions (every scored question right)",
              "points": [
                {
                  "label": "Claude Haiku 4.5 (thinking off) · Claude Code",
                  "value": 0.8659,
                  "lo": 0.7755,
                  "hi": 0.9234,
                  "n": 82,
                  "highlight": true
                },
                {
                  "label": "Claude Haiku 4.5 (thinking on) · Claude Code",
                  "value": 0.8902,
                  "lo": 0.8044,
                  "hi": 0.9412,
                  "n": 82,
                  "highlight": false
                },
                {
                  "label": "Claude Sonnet 5.5 (low) · Claude Code",
                  "value": 0.939,
                  "lo": 0.8651,
                  "hi": 0.9737,
                  "n": 82,
                  "highlight": false
                }
              ]
            },
            {
              "name": "Per-question accuracy",
              "points": [
                {
                  "label": "Claude Haiku 4.5 (thinking off) · Claude Code",
                  "value": 0.9124,
                  "lo": 0.8642,
                  "hi": 0.9446,
                  "n": 194,
                  "highlight": true
                },
                {
                  "label": "Claude Haiku 4.5 (thinking on) · Claude Code",
                  "value": 0.9433,
                  "lo": 0.9013,
                  "hi": 0.968,
                  "n": 194,
                  "highlight": false
                },
                {
                  "label": "Claude Sonnet 5.5 (low) · Claude Code",
                  "value": 0.9742,
                  "lo": 0.9411,
                  "hi": 0.9889,
                  "n": 194,
                  "highlight": false
                }
              ]
            }
          ],
          "note": "Whiskers are 95% Wilson intervals. Questions within a decision are related; per-question intervals are descriptive, not an independent-question test. The thinking-on and Sonnet arms are the recorded 2026-10-05 routing run, reused, not rerun; the thinking-off arm ran later on another account. An unanswered question counts as wrong.",
          "whisker": "ci95",
          "factContext": "typed routing decisions, thinking on vs off",
          "sourceIds": [
            "agent-haiku-thinking",
            "agent-routing"
          ]
        },
        {
          "id": "haiku-thinking-router-latency",
          "title": "Haiku thinking study: time per routing decision",
          "subtitle": "Median wall time and model (API) time; whisker to the 95th percentile",
          "kind": "dot-range",
          "unit": "seconds",
          "yLabel": "Seconds",
          "series": [
            {
              "name": "Wall time (CLI)",
              "points": [
                {
                  "label": "Claude Haiku 4.5 (thinking off) · Claude Code",
                  "value": 4.66,
                  "lo": 4.66,
                  "hi": 8.18,
                  "n": 82,
                  "highlight": true
                },
                {
                  "label": "Claude Haiku 4.5 (thinking on) · Claude Code",
                  "value": 12.54,
                  "lo": 12.54,
                  "hi": 34.48,
                  "n": 82,
                  "highlight": false
                },
                {
                  "label": "Claude Sonnet 5.5 (low) · Claude Code",
                  "value": 2.6,
                  "lo": 2.6,
                  "hi": 4.3,
                  "n": 82,
                  "highlight": false
                }
              ]
            },
            {
              "name": "Model time (API)",
              "points": [
                {
                  "label": "Claude Haiku 4.5 (thinking off) · Claude Code",
                  "value": 3.79,
                  "lo": 3.79,
                  "hi": 7.43,
                  "n": 82,
                  "highlight": true
                },
                {
                  "label": "Claude Haiku 4.5 (thinking on) · Claude Code",
                  "value": 10.51,
                  "lo": 10.51,
                  "hi": 32.13,
                  "n": 82,
                  "highlight": false
                },
                {
                  "label": "Claude Sonnet 5.5 (low) · Claude Code",
                  "value": 1.6,
                  "lo": 1.6,
                  "hi": 2.58,
                  "n": 82,
                  "highlight": false
                }
              ]
            }
          ],
          "note": "Whiskers run from p50 to p95, not a confidence interval. Percentiles use the nearest-rank rule of the routing-overhead study, so the thinking-on and Sonnet medians match that study (the Jev-vs-LLM page interpolates between ranks and shows slightly different values). One call at a time through the Claude Code CLI; the thinking-on and Sonnet arms ran on another day.",
          "whisker": "p50-p95",
          "factContext": "typed routing decisions, thinking on vs off",
          "sourceIds": [
            "agent-haiku-thinking",
            "agent-routing"
          ]
        },
        {
          "id": "haiku-thinking-router-tokens",
          "title": "Haiku thinking study: thinking and visible output tokens per routing decision",
          "subtitle": "Mean per decision, as the Claude Code CLI reports them",
          "kind": "grouped-bar",
          "unit": "tokens",
          "yLabel": "Tokens per decision",
          "series": [
            {
              "name": "Thinking tokens",
              "points": [
                {
                  "label": "Claude Haiku 4.5 (thinking off) · Claude Code",
                  "value": 0,
                  "n": 82
                },
                {
                  "label": "Claude Haiku 4.5 (thinking on) · Claude Code",
                  "value": 1101,
                  "n": 82
                },
                {
                  "label": "Claude Sonnet 5.5 (low) · Claude Code",
                  "value": 2,
                  "n": 82
                }
              ]
            },
            {
              "name": "Visible output tokens",
              "points": [
                {
                  "label": "Claude Haiku 4.5 (thinking off) · Claude Code",
                  "value": 366,
                  "n": 82
                },
                {
                  "label": "Claude Haiku 4.5 (thinking on) · Claude Code",
                  "value": 318,
                  "n": 82
                },
                {
                  "label": "Claude Sonnet 5.5 (low) · Claude Code",
                  "value": 105,
                  "n": 82
                }
              ]
            }
          ],
          "note": "Visible output = output tokens minus thinking tokens. It includes the structured answer the CLI asks for. Thinking tokens are counted by the CLI; their content is never captured. More tokens is not better or worse by itself.",
          "factContext": "typed routing decisions, thinking on vs off",
          "sourceIds": [
            "agent-haiku-thinking",
            "agent-routing"
          ]
        },
        {
          "id": "haiku-thinking-router-cost",
          "title": "Haiku thinking study: list-price cost per 1,000 routing decisions (calculation)",
          "subtitle": "Reported tokens × list price, per 1,000 decisions",
          "kind": "bar",
          "unit": "usd",
          "yLabel": "USD per 1,000 decisions",
          "series": [
            {
              "name": "Cost per 1,000 decisions",
              "points": [
                {
                  "label": "Claude Haiku 4.5 (thinking off) · Claude Code",
                  "value": 3.364,
                  "n": 82,
                  "highlight": true
                },
                {
                  "label": "Claude Haiku 4.5 (thinking on) · Claude Code",
                  "value": 8.924,
                  "n": 82,
                  "highlight": false
                },
                {
                  "label": "Claude Sonnet 5.5 (low) · Claude Code",
                  "value": 7.324,
                  "n": 82,
                  "highlight": false
                }
              ]
            }
          ],
          "note": "Calculation, not a bill: the calls ran on a flat subscription. Reported input, cache and output tokens (thinking tokens are part of output) × list price, with the price table of the Jev-vs-LLM study. Sonnet cache writes use the one-hour rate ($4 per million tokens), as recorded in its receipts. The calculation matches the CLI-reported cost.",
          "factContext": "typed routing decisions, thinking on vs off",
          "sourceIds": [
            "agent-haiku-thinking",
            "agent-routing",
            "calc-repricing",
            "price-anthropic"
          ]
        },
        {
          "id": "haiku-thinking-hard-pass",
          "title": "Haiku thinking study: pass rate on eight hard tasks",
          "subtitle": "Claude Haiku 4.5 in Claude Code; strict and lenient",
          "kind": "dot-range",
          "unit": "rate",
          "polarity": "higher",
          "yLabel": "Passed",
          "series": [
            {
              "name": "Strict pass",
              "points": [
                {
                  "label": "Claude Haiku 4.5 (thinking off) · Claude Code",
                  "value": 0.1667,
                  "lo": 0.0668,
                  "hi": 0.3585,
                  "n": 24,
                  "highlight": true
                },
                {
                  "label": "Claude Haiku 4.5 (thinking on) · Claude Code",
                  "value": 0.4583,
                  "lo": 0.2789,
                  "hi": 0.6493,
                  "n": 24,
                  "highlight": false
                }
              ]
            },
            {
              "name": "Lenient (format misses counted)",
              "points": [
                {
                  "label": "Claude Haiku 4.5 (thinking off) · Claude Code",
                  "value": 0.1667,
                  "lo": 0.0668,
                  "hi": 0.3585,
                  "n": 24,
                  "highlight": true
                },
                {
                  "label": "Claude Haiku 4.5 (thinking on) · Claude Code",
                  "value": 0.6667,
                  "lo": 0.4671,
                  "hi": 0.8203,
                  "n": 24,
                  "highlight": false
                }
              ]
            }
          ],
          "note": "Whiskers are 95% Wilson intervals over calls. Three repeats per task are related, so these intervals do not measure uncertainty across unseen tasks. A format miss (right answer in the wrong wrapping) never counts as a strict pass. The thinking-on cell is the 24 recorded Haiku receipts of the hard head-to-head, reused; the thinking-off cell ran later on another account.",
          "whisker": "ci95",
          "factContext": "eight hard validated tasks, thinking on vs off",
          "sourceIds": [
            "agent-haiku-thinking",
            "agent-provider-h2h-hard"
          ]
        },
        {
          "id": "haiku-thinking-hard-time",
          "title": "Haiku thinking study: total time per call on hard tasks",
          "subtitle": "Median per cell; whiskers = fastest and slowest call",
          "kind": "dot-range",
          "unit": "seconds",
          "yLabel": "Seconds",
          "series": [
            {
              "name": "Total time per call",
              "points": [
                {
                  "label": "Claude Haiku 4.5 (thinking off) · Claude Code",
                  "value": 2.95,
                  "lo": 1.7,
                  "hi": 13.01,
                  "n": 24,
                  "highlight": true
                },
                {
                  "label": "Claude Haiku 4.5 (thinking on) · Claude Code",
                  "value": 39.01,
                  "lo": 15.27,
                  "hi": 75.13,
                  "n": 24,
                  "highlight": false
                }
              ]
            }
          ],
          "note": "Whiskers are a range (fastest and slowest call), not a confidence interval. One host, one network; the thinking-on calls ran in an earlier session. Timings include CLI start-up.",
          "whisker": "minmax",
          "factContext": "eight hard validated tasks, thinking on vs off",
          "sourceIds": [
            "agent-haiku-thinking",
            "agent-provider-h2h-hard"
          ]
        }
      ],
      "tables": [
        {
          "id": "haiku-thinking-paired-test",
          "title": "Same cases, two arms: paired test on exact decisions",
          "columns": [
            {
              "key": "pair",
              "label": "Pair (first arm = thinking off)",
              "unit": "text"
            },
            {
              "key": "cases",
              "label": "Cases",
              "unit": "count"
            },
            {
              "key": "bothRight",
              "label": "Both right",
              "unit": "count"
            },
            {
              "key": "onlyOff",
              "label": "Only thinking off right",
              "unit": "count"
            },
            {
              "key": "onlyOther",
              "label": "Only the other arm right",
              "unit": "count"
            },
            {
              "key": "bothWrong",
              "label": "Both wrong",
              "unit": "count"
            },
            {
              "key": "p",
              "label": "Exact McNemar p (two-sided)"
            }
          ],
          "rows": [
            {
              "pair": "Thinking off vs thinking on (the same Haiku 4.5)",
              "cases": 82,
              "bothRight": 67,
              "onlyOff": 4,
              "onlyOther": 6,
              "bothWrong": 5,
              "p": 0.754
            },
            {
              "pair": "Thinking off vs Sonnet 5.5 at low effort (secondary)",
              "cases": 82,
              "bothRight": 70,
              "onlyOff": 1,
              "onlyOther": 7,
              "bothWrong": 4,
              "p": 0.07
            }
          ]
        },
        {
          "id": "haiku-thinking-router-cells",
          "title": "Routing: every arm",
          "columns": [
            {
              "key": "arm",
              "label": "Arm",
              "unit": "text"
            },
            {
              "key": "setting",
              "label": "Thinking setting",
              "unit": "text"
            },
            {
              "key": "calls",
              "label": "Calls (errors)",
              "unit": "text"
            },
            {
              "key": "exact",
              "label": "Exact",
              "unit": "text"
            },
            {
              "key": "keys",
              "label": "Per-question",
              "unit": "text"
            },
            {
              "key": "wall",
              "label": "Wall p50 / p95 (s)",
              "unit": "text"
            },
            {
              "key": "api",
              "label": "API p50 / p95 (s)",
              "unit": "text"
            },
            {
              "key": "wallRange",
              "label": "Wall min to max (s; not an interval)",
              "unit": "text"
            },
            {
              "key": "apiRange",
              "label": "API min to max (s; not an interval)",
              "unit": "text"
            },
            {
              "key": "thinking",
              "label": "Thinking tokens per decision (range)",
              "unit": "text"
            },
            {
              "key": "visible",
              "label": "Visible output tokens",
              "unit": "tokens"
            },
            {
              "key": "cost",
              "label": "USD per 1,000 (calculation)",
              "unit": "usd"
            }
          ],
          "rows": [
            {
              "arm": "Claude Haiku 4.5 (thinking off) · Claude Code",
              "setting": "MAX_THINKING_TOKENS=0",
              "calls": "82 (0)",
              "exact": "87% (71/82; 95% interval 78% to 92%)",
              "keys": "91% (177/194; 95% interval 86% to 94%)",
              "wall": "4.66 / 8.18",
              "api": "3.79 / 7.43",
              "wallRange": "2.2 to 11.57",
              "apiRange": "1.37 to 10.83",
              "thinking": "0 (0 to 0)",
              "visible": 366,
              "cost": 3.364
            },
            {
              "arm": "Claude Haiku 4.5 (thinking on) · Claude Code",
              "setting": "CLI default",
              "calls": "82 (0)",
              "exact": "89% (73/82; 95% interval 80% to 94%)",
              "keys": "94% (183/194; 95% interval 90% to 97%)",
              "wall": "12.54 / 34.48",
              "api": "10.51 / 32.13",
              "wallRange": "5.86 to 51.28",
              "apiRange": "4.33 to 49.49",
              "thinking": "1101 (285 to 4443)",
              "visible": 318,
              "cost": 8.924
            },
            {
              "arm": "Claude Sonnet 5.5 (low) · Claude Code",
              "setting": "--effort low",
              "calls": "82 (0)",
              "exact": "94% (77/82; 95% interval 87% to 97%)",
              "keys": "97% (189/194; 95% interval 94% to 99%)",
              "wall": "2.6 / 4.3",
              "api": "1.6 / 2.58",
              "wallRange": "1.99 to 5.58",
              "apiRange": "1.06 to 4.76",
              "thinking": "2 (0 to 63)",
              "visible": 105,
              "cost": 7.324
            }
          ]
        },
        {
          "id": "haiku-thinking-by-decision-type",
          "title": "Routing: exact decisions by decision type",
          "columns": [
            {
              "key": "decision",
              "label": "Decision type (cases)",
              "unit": "text"
            },
            {
              "key": "off",
              "label": "Claude Haiku 4.5 (thinking off) · Claude Code",
              "unit": "text"
            },
            {
              "key": "on",
              "label": "Claude Haiku 4.5 (thinking on) · Claude Code",
              "unit": "text"
            },
            {
              "key": "sonnet",
              "label": "Claude Sonnet 5.5 (low) · Claude Code",
              "unit": "text"
            }
          ],
          "rows": [
            {
              "decision": "Failure class (18)",
              "off": "83% (15/18; 95% interval 61% to 94%)",
              "on": "94% (17/18; 95% interval 74% to 99%)",
              "sonnet": "100% (18/18; 95% interval 82% to 100%)"
            },
            {
              "decision": "Message intent (20)",
              "off": "100% (20/20; 95% interval 84% to 100%)",
              "on": "100% (20/20; 95% interval 84% to 100%)",
              "sonnet": "100% (20/20; 95% interval 84% to 100%)"
            },
            {
              "decision": "Is it a rule? (12)",
              "off": "100% (12/12; 95% interval 76% to 100%)",
              "on": "100% (12/12; 95% interval 76% to 100%)",
              "sonnet": "100% (12/12; 95% interval 76% to 100%)"
            },
            {
              "decision": "Context shape (32)",
              "off": "75% (24/32; 95% interval 58% to 87%)",
              "on": "75% (24/32; 95% interval 58% to 87%)",
              "sonnet": "84% (27/32; 95% interval 68% to 93%)"
            }
          ]
        },
        {
          "id": "haiku-thinking-hard-cells",
          "title": "Hard tasks: both cells",
          "columns": [
            {
              "key": "cell",
              "label": "Cell",
              "unit": "text"
            },
            {
              "key": "calls",
              "label": "Calls",
              "unit": "count"
            },
            {
              "key": "strict",
              "label": "Strict pass",
              "unit": "text"
            },
            {
              "key": "formatMisses",
              "label": "Format misses",
              "unit": "count"
            },
            {
              "key": "wrong",
              "label": "Wrong answers",
              "unit": "count"
            },
            {
              "key": "noAnswer",
              "label": "Errors or incomplete calls (failures)",
              "unit": "count"
            },
            {
              "key": "total",
              "label": "Total time median (min to max, s)",
              "unit": "text"
            },
            {
              "key": "reasoning",
              "label": "Reasoning tokens median",
              "unit": "tokens"
            },
            {
              "key": "output",
              "label": "Output tokens median",
              "unit": "tokens"
            },
            {
              "key": "perPass",
              "label": "USD per strict pass (calculation)",
              "unit": "usd"
            }
          ],
          "rows": [
            {
              "cell": "Claude Haiku 4.5 (thinking off) · Claude Code",
              "calls": 24,
              "strict": "17% (4/24; 95% interval 7% to 36%)",
              "formatMisses": 0,
              "wrong": 20,
              "noAnswer": 0,
              "total": "2.9 (1.7 to 13)",
              "reasoning": 0,
              "output": 291,
              "perPass": 0.03654
            },
            {
              "cell": "Claude Haiku 4.5 (thinking on) · Claude Code",
              "calls": 24,
              "strict": "46% (11/24; 95% interval 28% to 65%)",
              "formatMisses": 5,
              "wrong": 8,
              "noAnswer": 0,
              "total": "39 (15.3 to 75.1)",
              "reasoning": 4556,
              "output": 5064,
              "perPass": 0.0672
            }
          ]
        },
        {
          "id": "haiku-thinking-hard-matrix",
          "title": "Hard tasks: strict passes per task",
          "columns": [
            {
              "key": "task",
              "label": "Task",
              "unit": "text"
            },
            {
              "key": "off",
              "label": "Claude Haiku 4.5 (thinking off) · Claude Code",
              "unit": "text"
            },
            {
              "key": "on",
              "label": "Claude Haiku 4.5 (thinking on) · Claude Code",
              "unit": "text"
            }
          ],
          "rows": [
            {
              "task": "Fix an interval-merge function (off-by-one and edge cases)",
              "off": "100% (3/3; 95% interval 44% to 100%)",
              "on": "100% (3/3; 95% interval 44% to 100%)"
            },
            {
              "task": "Fix a time-zone day-length function (DST)",
              "off": "0% (0/3; 95% interval 0% to 56%)",
              "on": "33% (1/3; 95% interval 6% to 79%)"
            },
            {
              "task": "Write a CSV parser (quoted newlines, strict errors)",
              "off": "0% (0/3; 95% interval 0% to 56%)",
              "on": "67% (2/3; 95% interval 21% to 94%) (+1 format miss)"
            },
            {
              "task": "Predict JavaScript event-loop output order",
              "off": "0% (0/3; 95% interval 0% to 56%)",
              "on": "0% (0/3; 95% interval 0% to 56%)"
            },
            {
              "task": "Solve a multi-constraint room schedule",
              "off": "0% (0/3; 95% interval 0% to 56%)",
              "on": "0% (0/3; 95% interval 0% to 56%) (+2 format misses)"
            },
            {
              "task": "Write a strict SemVer 2.0.0 regex",
              "off": "0% (0/3; 95% interval 0% to 56%)",
              "on": "100% (3/3; 95% interval 44% to 100%)"
            },
            {
              "task": "Refactor to remove duplication, keep 20 tests green",
              "off": "33% (1/3; 95% interval 6% to 79%)",
              "on": "67% (2/3; 95% interval 21% to 94%)"
            },
            {
              "task": "Write a SQLite reporting query (fan-out, ties, boundaries)",
              "off": "0% (0/3; 95% interval 0% to 56%)",
              "on": "0% (0/3; 95% interval 0% to 56%) (+2 format misses)"
            }
          ]
        }
      ],
      "related": [
        "routing-jev-vs-llm",
        "hard-model-head-to-head",
        "routing-overhead"
      ]
    },
    {
      "slug": "json-schema-vs-instructions",
      "title": "Does a JSON schema stop format misses? Instructions vs schema mode in Claude Code and Codex CLI",
      "seoTitle": "LLM structured output: JSON schema vs instructions",
      "description": "96 calls. Strict passes, schema vs instructions: Haiku 18/24 vs 0/24, Sonnet 12/12 vs 12/12, GPT-6.1 Sol 12/12 vs 12/12. With intervals.",
      "question": "For the same extraction tasks, does enforcing a JSON schema through the CLI change the strict pass rate, the format misses and the wrong values, compared with asking for JSON in the prompt?",
      "answer": "96 counted calls on three JSON extraction prompts, each asked with instructions only and with the CLI’s JSON schema mode. Format misses (a right answer in the wrong format): 17 of 48 calls with instructions (95% interval 23% to 50%), 0 of 48 with a schema (95% interval 0% to 7%). Code fences: 24/48 instruction replies (95% interval 36% to 64%). Schema replies: 0/48 (95% interval 0% to 7%). The instructions said not to use a fence. Strict passes: Haiku: 0/24 (95% interval 0% to 14%; 17 format misses; 7 wrong-value replies) with instructions and 18/24 (95% interval 55% to 88%; 6 wrong-value replies) with a schema; Sonnet: 12/12 (95% interval 76% to 100%) with instructions and 12/12 (95% interval 76% to 100%) with a schema; GPT-6.1 Sol: 12/12 (95% interval 76% to 100%) with instructions and 12/12 (95% interval 76% to 100%) with a schema. Wrong values: 7 of 48 with instructions (95% interval 7% to 27%), 6 of 48 with a schema (95% interval 6% to 25%). These schemas constrain the reply’s shape. They do not verify totals or due dates. Where the 95% intervals do not overlap: Haiku, schema ahead (18/24 vs 0/24; paired McNemar p = 0.00000762939453125, calculation); every other pair overlaps, so the data does not rank it. Sonnet and GPT-6.1 Sol passed every call in both modes, a ceiling on this task set that cannot show a schema effect. Recommendation: test the CLI’s schema mode on your own tasks. On this task set, use it for Haiku, where strict passes were ahead (the 95% intervals do not overlap); for Sonnet and GPT-6.1 Sol the intervals overlap, so this sample does not establish a difference. Keep a validator for totals and dates; these schemas do not check them.",
      "date": "2026-10-07",
      "updated": "2026-10-07",
      "tags": [
        "structured-output",
        "json-schema",
        "format-miss",
        "claude-code",
        "codex-cli",
        "claude-haiku",
        "claude-sonnet",
        "gpt-6-1-sol"
      ],
      "method": [
        "The protocol states a pre-call declaration. Its current file birth time is 2026-10-07T01:47:06.955Z. The first counted call began at 2026-10-07T00:43:42.104Z. These files cannot verify a protocol declaration before the Codex calls. The extract keeps every counted call.",
        "Three prompts each have one exact expected JSON answer and a sandboxed validator. The order task merges lines, drops a cancelled line and applies a discount. The invoice task applies a line discount and then tax, rounded half up. It also includes an untaxed fee and shipping. The meeting task extracts tickets, owners and due dates, resolves relative dates, and skips a decision and a note. We do not publish prompts or replies.",
        "Two modes use the same task text. Instructions mode adds “Reply with the JSON object only, no code fence, no other text.” Schema mode removes that sentence and uses the CLI’s schema flag. Claude Code uses --json-schema; the harness reads its structured_output field. Codex exec uses --output-schema; the harness reads its final message.",
        "Repetitions per prompt, in each mode: Haiku: 8; Sonnet: 4; GPT-6.1 Sol: 4. The order alternates by repetition, so time drift falls on both modes. Claude rows use the CLI’s default effort. GPT-6.1 Sol rows use low effort.",
        "Scoring: a strict pass needs the whole reply to be the right JSON. A format miss is a right answer inside a code fence or prose. Wrong values means any other completed reply, including a fenced reply with a wrong value. An error means the call did not complete; it counts as a fail.",
        "Rates carry Wilson 95% intervals (a calculation). A side is ahead only when the intervals do not overlap and the paired exact McNemar p is below 0.05. The paired table gives this p-value calculation. Time is a median with the fastest and slowest call: a range, not an interval.",
        "Controls ran before inference. Each reference answer passes, including through the schema path. A reference inside a code fence counts as a format miss. Four planted wrong answers per prompt fail. One uncounted probe per route (2 here) checked how the harness reads the CLI output. Probes stay outside every cell.",
        "Each call uses a fresh empty working folder, with tools off, no MCP servers and no session persistence. The harness reads the answer from the CLI output. Claude Code uses its JSON output format in both modes; earlier studies used its stream format. Schema mode adds the schema flag and removes the format sentence.",
        "Every counted batch completed all planned cells."
      ],
      "caveats": [
        "Small samples: Haiku 24 calls per mode, Sonnet 12 calls per mode and GPT-6.1 Sol 12 calls per mode. A 12/12 result has a 95% interval of 76% to 100%. This is not a minimum detectable difference.",
        "The calls repeat only three fixed prompts. Wilson intervals describe call outcomes under a binomial assumption; they do not measure accuracy across unseen tasks. The paired p-values also assume independent pairs and do not remove this limit.",
        "The prompts are hand-made. The order prompt reuses a case with known Haiku format misses from an earlier study; this is not a blind holdout.",
        "Both routes used one shared Mac. The files do not establish host isolation from other work. Timings include CLI start-up and network time.",
        "The current protocol file was born after the Codex batch. An earlier version may have existed, but these files cannot verify pre-call registration. Amendment 1 says 00:44 UTC and before counted calls; the Codex batch started at 00:43 UTC. Amendment 4 says 06:10 UTC and before the Claude probe; the probe receipt was saved at 06:04 UTC and counted Claude calls began at 06:04 UTC. These timing conflicts limit the protocol claims. The calls remain descriptive evidence.",
        "The first Claude lane stopped after a 90-minute wait for another study to release the route. It made no inference call. The lane restarted later; no counted call was retried.",
        "The comparison changes both the schema flag and the format sentence. It cannot isolate the effect of the flag alone.",
        "These schemas constrain keys, types and currency labels. They do not verify totals or due dates. Both modes use the same task text. A change in wrong values is a measured result, not a design aim.",
        "Sonnet and GPT-6.1 Sol passed every call in both modes. These three tasks are within reach of those models, so they cannot show a schema effect for them; harder extraction may differ.",
        "The instruction wording is one sentence; another wording, or a schema with the format sentence kept, was not tested. The format-miss counts here are not comparable with the caching and consistency study, which used a different wording on one of these prompts.",
        "In schema mode Claude Code returns the answer through a structured-output tool call. Its median was 2 model turns (range 2 to 2, n = 36). Instructions had a median of 1 (range 1 to 1, n = 36). Its time and token counts include that extra step.",
        "The Codex CLI’s exec mode does not echo the model or the reasoning effort. The request named GPT-6.1 Sol at low effort, and the CLI’s own catalog lists that model with that effort, but a reroute would not have been visible. Codex timings include its start-up and its larger system prompt, so Codex rows compare route and model pairs, not models alone."
      ],
      "sourceIds": [
        "agent-structured-output"
      ],
      "stats": [
        {
          "id": "structured-output-calls",
          "label": "Counted calls in this study (every one counted)",
          "value": 96,
          "unit": "calls",
          "display": "96 (72 Claude Code, 24 Codex CLI)"
        },
        {
          "id": "structured-output-format-miss-instructions",
          "label": "Format-miss rate with instructions only, all models",
          "value": 0.3542,
          "unit": "rate",
          "display": "35% (17/48)",
          "n": 48,
          "ci": [
            0.2343,
            0.4956
          ],
          "note": "A right answer in a code fence or prose. Every error counted as a call."
        },
        {
          "id": "structured-output-format-miss-schema",
          "label": "Format-miss rate with a JSON schema, all models",
          "value": 0,
          "unit": "rate",
          "display": "0% (0/48)",
          "n": 48,
          "ci": [
            0,
            0.0741
          ],
          "note": "A right answer in a code fence or prose. Every error counted as a call."
        },
        {
          "id": "structured-output-fenced-instructions",
          "label": "Replies in a code fence with instructions only, all models",
          "value": 0.5,
          "unit": "rate",
          "display": "50% (24/48)",
          "n": 48,
          "ci": [
            0.3639,
            0.6361
          ],
          "note": "The prompt said: no code fence. Completed replies only; a fence can hold a right or a wrong answer."
        },
        {
          "id": "structured-output-fenced-schema",
          "label": "Replies in a code fence with a JSON schema, all models",
          "value": 0,
          "unit": "rate",
          "display": "0% (0/48)",
          "n": 48,
          "ci": [
            0,
            0.0741
          ],
          "note": "Completed replies only. Claude Code uses its structured_output field. Codex CLI uses its final message."
        },
        {
          "id": "structured-output-wrong-values-instructions",
          "label": "Wrong-values rate with instructions only, all models",
          "value": 0.1458,
          "unit": "rate",
          "display": "15% (7/48)",
          "n": 48,
          "ci": [
            0.0725,
            0.2717
          ],
          "note": "A completed reply that is neither a strict pass nor a format miss. Denominator includes all attempts; errors are a separate outcome."
        },
        {
          "id": "structured-output-wrong-values-schema",
          "label": "Wrong-values rate with a JSON schema, all models",
          "value": 0.125,
          "unit": "rate",
          "display": "13% (6/48)",
          "n": 48,
          "ci": [
            0.0586,
            0.247
          ],
          "note": "A completed reply that is neither a strict pass nor a format miss. Denominator includes all attempts; errors are a separate outcome."
        },
        {
          "id": "structured-output-strict-instructions",
          "label": "Strict pass rate with instructions only, all models",
          "value": 0.5,
          "unit": "rate",
          "display": "50% (24/48)",
          "n": 48,
          "ci": [
            0.3639,
            0.6361
          ],
          "note": "Pooled over three models and three prompts; the models differ in n, so read the per-model chart."
        },
        {
          "id": "structured-output-strict-schema",
          "label": "Strict pass rate with a JSON schema, all models",
          "value": 0.875,
          "unit": "rate",
          "display": "88% (42/48)",
          "n": 48,
          "ci": [
            0.753,
            0.9414
          ],
          "note": "Pooled over three models and three prompts; the models differ in n, so read the per-model chart."
        }
      ],
      "charts": [
        {
          "id": "structured-output-pass-rate",
          "title": "Does a JSON schema raise the pass rate? Instructions vs schema mode",
          "subtitle": "Three extraction prompts pooled; whiskers are 95% Wilson intervals",
          "kind": "dot-range",
          "unit": "rate",
          "polarity": "higher",
          "yLabel": "Passed",
          "series": [
            {
              "name": "Strict pass: the whole reply is the right JSON",
              "points": [
                {
                  "label": "Claude Haiku 4.5 (instructions) · Claude Code",
                  "value": 0,
                  "lo": 0,
                  "hi": 0.138,
                  "n": 24
                },
                {
                  "label": "Claude Haiku 4.5 (JSON schema) · Claude Code",
                  "value": 0.75,
                  "lo": 0.551,
                  "hi": 0.88,
                  "n": 24
                },
                {
                  "label": "Claude Sonnet 5.5 (instructions) · Claude Code",
                  "value": 1,
                  "lo": 0.7575,
                  "hi": 1,
                  "n": 12
                },
                {
                  "label": "Claude Sonnet 5.5 (JSON schema) · Claude Code",
                  "value": 1,
                  "lo": 0.7575,
                  "hi": 1,
                  "n": 12
                },
                {
                  "label": "GPT-6.1 Sol (low, instructions) · Codex CLI",
                  "value": 1,
                  "lo": 0.7575,
                  "hi": 1,
                  "n": 12
                },
                {
                  "label": "GPT-6.1 Sol (low, JSON schema) · Codex CLI",
                  "value": 1,
                  "lo": 0.7575,
                  "hi": 1,
                  "n": 12
                }
              ]
            },
            {
              "name": "Right answer in any format (strict pass or format miss)",
              "points": [
                {
                  "label": "Claude Haiku 4.5 (instructions) · Claude Code",
                  "value": 0.7083,
                  "lo": 0.5083,
                  "hi": 0.8509,
                  "n": 24
                },
                {
                  "label": "Claude Haiku 4.5 (JSON schema) · Claude Code",
                  "value": 0.75,
                  "lo": 0.551,
                  "hi": 0.88,
                  "n": 24
                },
                {
                  "label": "Claude Sonnet 5.5 (instructions) · Claude Code",
                  "value": 1,
                  "lo": 0.7575,
                  "hi": 1,
                  "n": 12
                },
                {
                  "label": "Claude Sonnet 5.5 (JSON schema) · Claude Code",
                  "value": 1,
                  "lo": 0.7575,
                  "hi": 1,
                  "n": 12
                },
                {
                  "label": "GPT-6.1 Sol (low, instructions) · Codex CLI",
                  "value": 1,
                  "lo": 0.7575,
                  "hi": 1,
                  "n": 12
                },
                {
                  "label": "GPT-6.1 Sol (low, JSON schema) · Codex CLI",
                  "value": 1,
                  "lo": 0.7575,
                  "hi": 1,
                  "n": 12
                }
              ]
            }
          ],
          "note": "Whiskers are 95% Wilson intervals (a calculation) over 24 calls and 12 calls per configuration; every error counts as a fail. Strict: the whole reply parses as JSON and matches the expected answer exactly. A format miss is a right answer inside a code fence or prose, so it is never a strict pass.",
          "whisker": "ci95",
          "sourceIds": [
            "agent-structured-output"
          ]
        },
        {
          "id": "structured-output-outcomes",
          "title": "What each call produced: strict pass, format miss, wrong values or error",
          "subtitle": "Counts of calls per configuration; the three prompts pooled",
          "kind": "stacked-bar",
          "unit": "count",
          "yLabel": "Calls",
          "series": [
            {
              "name": "Strict pass",
              "points": [
                {
                  "label": "Claude Haiku 4.5 (instructions) · Claude Code",
                  "value": 0,
                  "n": 24
                },
                {
                  "label": "Claude Haiku 4.5 (JSON schema) · Claude Code",
                  "value": 18,
                  "n": 24
                },
                {
                  "label": "Claude Sonnet 5.5 (instructions) · Claude Code",
                  "value": 12,
                  "n": 12
                },
                {
                  "label": "Claude Sonnet 5.5 (JSON schema) · Claude Code",
                  "value": 12,
                  "n": 12
                },
                {
                  "label": "GPT-6.1 Sol (low, instructions) · Codex CLI",
                  "value": 12,
                  "n": 12
                },
                {
                  "label": "GPT-6.1 Sol (low, JSON schema) · Codex CLI",
                  "value": 12,
                  "n": 12
                }
              ]
            },
            {
              "name": "Format miss",
              "points": [
                {
                  "label": "Claude Haiku 4.5 (instructions) · Claude Code",
                  "value": 17,
                  "n": 24
                },
                {
                  "label": "Claude Haiku 4.5 (JSON schema) · Claude Code",
                  "value": 0,
                  "n": 24
                },
                {
                  "label": "Claude Sonnet 5.5 (instructions) · Claude Code",
                  "value": 0,
                  "n": 12
                },
                {
                  "label": "Claude Sonnet 5.5 (JSON schema) · Claude Code",
                  "value": 0,
                  "n": 12
                },
                {
                  "label": "GPT-6.1 Sol (low, instructions) · Codex CLI",
                  "value": 0,
                  "n": 12
                },
                {
                  "label": "GPT-6.1 Sol (low, JSON schema) · Codex CLI",
                  "value": 0,
                  "n": 12
                }
              ]
            },
            {
              "name": "Wrong values",
              "points": [
                {
                  "label": "Claude Haiku 4.5 (instructions) · Claude Code",
                  "value": 7,
                  "n": 24
                },
                {
                  "label": "Claude Haiku 4.5 (JSON schema) · Claude Code",
                  "value": 6,
                  "n": 24
                },
                {
                  "label": "Claude Sonnet 5.5 (instructions) · Claude Code",
                  "value": 0,
                  "n": 12
                },
                {
                  "label": "Claude Sonnet 5.5 (JSON schema) · Claude Code",
                  "value": 0,
                  "n": 12
                },
                {
                  "label": "GPT-6.1 Sol (low, instructions) · Codex CLI",
                  "value": 0,
                  "n": 12
                },
                {
                  "label": "GPT-6.1 Sol (low, JSON schema) · Codex CLI",
                  "value": 0,
                  "n": 12
                }
              ]
            },
            {
              "name": "Error",
              "points": [
                {
                  "label": "Claude Haiku 4.5 (instructions) · Claude Code",
                  "value": 0,
                  "n": 24
                },
                {
                  "label": "Claude Haiku 4.5 (JSON schema) · Claude Code",
                  "value": 0,
                  "n": 24
                },
                {
                  "label": "Claude Sonnet 5.5 (instructions) · Claude Code",
                  "value": 0,
                  "n": 12
                },
                {
                  "label": "Claude Sonnet 5.5 (JSON schema) · Claude Code",
                  "value": 0,
                  "n": 12
                },
                {
                  "label": "GPT-6.1 Sol (low, instructions) · Codex CLI",
                  "value": 0,
                  "n": 12
                },
                {
                  "label": "GPT-6.1 Sol (low, JSON schema) · Codex CLI",
                  "value": 0,
                  "n": 12
                }
              ]
            }
          ],
          "note": "Counts of calls, not rates; the pass-rate chart carries the same results with 95% intervals. Format miss: the right answer inside a code fence or prose. Wrong values: any other completed reply, with a wrong value, key or type (a reply that sits in a code fence and also has a wrong value is counted here). Error: the call did not complete.",
          "sourceIds": [
            "agent-structured-output"
          ]
        },
        {
          "id": "structured-output-time",
          "title": "Time per call, instructions vs schema mode",
          "subtitle": "Median; whiskers = fastest and slowest completed call",
          "kind": "dot-range",
          "unit": "seconds",
          "yLabel": "Seconds",
          "series": [
            {
              "name": "Median time per call (the three prompts pooled)",
              "points": [
                {
                  "label": "Claude Haiku 4.5 (instructions) · Claude Code",
                  "value": 9.52,
                  "lo": 5.67,
                  "hi": 17,
                  "n": 24
                },
                {
                  "label": "Claude Haiku 4.5 (JSON schema) · Claude Code",
                  "value": 8.46,
                  "lo": 5.9,
                  "hi": 12.23,
                  "n": 24
                },
                {
                  "label": "Claude Sonnet 5.5 (instructions) · Claude Code",
                  "value": 3.52,
                  "lo": 2.67,
                  "hi": 4.12,
                  "n": 12
                },
                {
                  "label": "Claude Sonnet 5.5 (JSON schema) · Claude Code",
                  "value": 4.2,
                  "lo": 2.95,
                  "hi": 6.14,
                  "n": 12
                },
                {
                  "label": "GPT-6.1 Sol (low, instructions) · Codex CLI",
                  "value": 6.21,
                  "lo": 4.2,
                  "hi": 12.27,
                  "n": 12
                },
                {
                  "label": "GPT-6.1 Sol (low, JSON schema) · Codex CLI",
                  "value": 5.96,
                  "lo": 4.62,
                  "hi": 20.97,
                  "n": 12
                }
              ]
            }
          ],
          "note": "Whiskers are a range (fastest and slowest call), not a confidence interval. Wall time from process start to exit, so it includes CLI start-up; Codex CLI timings include its larger system prompt. The three prompts differ in length, which widens every range.",
          "whisker": "minmax",
          "sourceIds": [
            "agent-structured-output"
          ]
        },
        {
          "id": "structured-output-tokens",
          "title": "Output and reasoning tokens per call, instructions vs schema mode",
          "subtitle": "Median per call; ranges and sample sizes are in the note",
          "kind": "grouped-bar",
          "unit": "tokens",
          "yLabel": "Tokens",
          "series": [
            {
              "name": "Median output tokens per call",
              "points": [
                {
                  "label": "Claude Haiku 4.5 (instructions) · Claude Code",
                  "value": 1128,
                  "n": 24
                },
                {
                  "label": "Claude Haiku 4.5 (JSON schema) · Claude Code",
                  "value": 1036,
                  "n": 24
                },
                {
                  "label": "Claude Sonnet 5.5 (instructions) · Claude Code",
                  "value": 368,
                  "n": 12
                },
                {
                  "label": "Claude Sonnet 5.5 (JSON schema) · Claude Code",
                  "value": 424,
                  "n": 12
                },
                {
                  "label": "GPT-6.1 Sol (low, instructions) · Codex CLI",
                  "value": 117,
                  "n": 12
                },
                {
                  "label": "GPT-6.1 Sol (low, JSON schema) · Codex CLI",
                  "value": 123,
                  "n": 12
                }
              ]
            },
            {
              "name": "Median reasoning tokens per call (thinking)",
              "points": [
                {
                  "label": "Claude Haiku 4.5 (instructions) · Claude Code",
                  "value": 934,
                  "n": 24
                },
                {
                  "label": "Claude Haiku 4.5 (JSON schema) · Claude Code",
                  "value": 727,
                  "n": 24
                },
                {
                  "label": "Claude Sonnet 5.5 (instructions) · Claude Code",
                  "value": 182,
                  "n": 12
                },
                {
                  "label": "Claude Sonnet 5.5 (JSON schema) · Claude Code",
                  "value": 109,
                  "n": 12
                },
                {
                  "label": "GPT-6.1 Sol (low, instructions) · Codex CLI",
                  "value": 21,
                  "n": 12
                },
                {
                  "label": "GPT-6.1 Sol (low, JSON schema) · Codex CLI",
                  "value": 23,
                  "n": 12
                }
              ]
            }
          ],
          "note": "Output tokens include reasoning tokens. Claude schema-mode output includes the CLI’s structured-output tool call. Input counts include CLI context and are not compared. Ranges (not intervals): Claude Haiku 4.5 (instructions) · Claude Code: output 698 to 2059 (n = 24); reasoning 559 to 1830 (n = 24); Claude Haiku 4.5 (JSON schema) · Claude Code: output 716 to 1462 (n = 24); reasoning 522 to 1187 (n = 24); Claude Sonnet 5.5 (instructions) · Claude Code: output 154 to 456 (n = 12); reasoning 54 to 278 (n = 12); Claude Sonnet 5.5 (JSON schema) · Claude Code: output 273 to 541 (n = 12); reasoning 0 to 260 (n = 12); GPT-6.1 Sol (low, instructions) · Codex CLI: output 69 to 259 (n = 12); reasoning 0 to 60 (n = 12); GPT-6.1 Sol (low, JSON schema) · Codex CLI: output 98 to 181 (n = 12); reasoning 0 to 49 (n = 12).",
          "sourceIds": [
            "agent-structured-output"
          ]
        }
      ],
      "tables": [
        {
          "id": "structured-output-cells",
          "title": "Every cell: configuration by prompt",
          "columns": [
            {
              "key": "config",
              "label": "Configuration",
              "unit": "text"
            },
            {
              "key": "prompt",
              "label": "Prompt",
              "unit": "text"
            },
            {
              "key": "strict",
              "label": "Strict passes",
              "unit": "text"
            },
            {
              "key": "ci",
              "label": "95% interval",
              "unit": "text"
            },
            {
              "key": "formatMisses",
              "label": "Format misses",
              "unit": "count"
            },
            {
              "key": "wrong",
              "label": "Wrong values",
              "unit": "count"
            },
            {
              "key": "errors",
              "label": "Errors",
              "unit": "count"
            },
            {
              "key": "fenced",
              "label": "Replies in a code fence",
              "unit": "count"
            },
            {
              "key": "completedN",
              "label": "Completed calls (n for time and tokens)",
              "unit": "count"
            },
            {
              "key": "medianTotal",
              "label": "Median time (s)",
              "unit": "seconds"
            },
            {
              "key": "timeRange",
              "label": "Time range (s, not an interval)",
              "unit": "text"
            },
            {
              "key": "medianOutput",
              "label": "Median output tokens",
              "unit": "tokens"
            },
            {
              "key": "outputRange",
              "label": "Output token range",
              "unit": "text"
            },
            {
              "key": "medianReasoning",
              "label": "Median reasoning tokens",
              "unit": "tokens"
            },
            {
              "key": "reasoningRange",
              "label": "Reasoning token range",
              "unit": "text"
            }
          ],
          "rows": [
            {
              "config": "Claude Haiku 4.5 (instructions) · Claude Code",
              "prompt": "Order to JSON",
              "strict": "0/8",
              "ci": "0% to 32%",
              "formatMisses": 8,
              "wrong": 0,
              "errors": 0,
              "fenced": 8,
              "medianTotal": 7.01,
              "timeRange": "5.67 to 9.8",
              "medianOutput": 897,
              "outputRange": "698 to 1134",
              "medianReasoning": 758.5,
              "reasoningRange": "559 to 996",
              "completedN": 8
            },
            {
              "config": "Claude Haiku 4.5 (instructions) · Claude Code",
              "prompt": "Invoice to JSON",
              "strict": "0/8",
              "ci": "0% to 32%",
              "formatMisses": 8,
              "wrong": 0,
              "errors": 0,
              "fenced": 8,
              "medianTotal": 9.92,
              "timeRange": "7.19 to 13.03",
              "medianOutput": 1280,
              "outputRange": "1088 to 1636",
              "medianReasoning": 1047,
              "reasoningRange": "856 to 1403",
              "completedN": 8
            },
            {
              "config": "Claude Haiku 4.5 (instructions) · Claude Code",
              "prompt": "Meeting notes to action items",
              "strict": "0/8",
              "ci": "0% to 32%",
              "formatMisses": 1,
              "wrong": 7,
              "errors": 0,
              "fenced": 8,
              "medianTotal": 10.34,
              "timeRange": "8.2 to 17",
              "medianOutput": 1235,
              "outputRange": "991 to 2059",
              "medianReasoning": 1030,
              "reasoningRange": "761 to 1830",
              "completedN": 8
            },
            {
              "config": "Claude Haiku 4.5 (JSON schema) · Claude Code",
              "prompt": "Order to JSON",
              "strict": "8/8",
              "ci": "68% to 100%",
              "formatMisses": 0,
              "wrong": 0,
              "errors": 0,
              "fenced": 0,
              "medianTotal": 7.19,
              "timeRange": "6.45 to 10.5",
              "medianOutput": 791.5,
              "outputRange": "716 to 1222",
              "medianReasoning": 586,
              "reasoningRange": "522 to 1010",
              "completedN": 8
            },
            {
              "config": "Claude Haiku 4.5 (JSON schema) · Claude Code",
              "prompt": "Invoice to JSON",
              "strict": "8/8",
              "ci": "68% to 100%",
              "formatMisses": 0,
              "wrong": 0,
              "errors": 0,
              "fenced": 0,
              "medianTotal": 9.39,
              "timeRange": "7.7 to 9.91",
              "medianOutput": 1120.5,
              "outputRange": "909 to 1226",
              "medianReasoning": 788,
              "reasoningRange": "576 to 893",
              "completedN": 8
            },
            {
              "config": "Claude Haiku 4.5 (JSON schema) · Claude Code",
              "prompt": "Meeting notes to action items",
              "strict": "2/8",
              "ci": "7% to 59%",
              "formatMisses": 0,
              "wrong": 6,
              "errors": 0,
              "fenced": 0,
              "medianTotal": 9.27,
              "timeRange": "5.9 to 12.23",
              "medianOutput": 1164.5,
              "outputRange": "806 to 1462",
              "medianReasoning": 890.5,
              "reasoningRange": "532 to 1187",
              "completedN": 8
            },
            {
              "config": "Claude Sonnet 5.5 (instructions) · Claude Code",
              "prompt": "Order to JSON",
              "strict": "4/4",
              "ci": "51% to 100%",
              "formatMisses": 0,
              "wrong": 0,
              "errors": 0,
              "fenced": 0,
              "medianTotal": 3.27,
              "timeRange": "2.67 to 3.8",
              "medianOutput": 166.5,
              "outputRange": "154 to 175",
              "medianReasoning": 66.5,
              "reasoningRange": "54 to 75",
              "completedN": 4
            },
            {
              "config": "Claude Sonnet 5.5 (instructions) · Claude Code",
              "prompt": "Invoice to JSON",
              "strict": "4/4",
              "ci": "51% to 100%",
              "formatMisses": 0,
              "wrong": 0,
              "errors": 0,
              "fenced": 0,
              "medianTotal": 3.65,
              "timeRange": "3.32 to 4.12",
              "medianOutput": 367.5,
              "outputRange": "362 to 373",
              "medianReasoning": 181.5,
              "reasoningRange": "176 to 187",
              "completedN": 4
            },
            {
              "config": "Claude Sonnet 5.5 (instructions) · Claude Code",
              "prompt": "Meeting notes to action items",
              "strict": "4/4",
              "ci": "51% to 100%",
              "formatMisses": 0,
              "wrong": 0,
              "errors": 0,
              "fenced": 0,
              "medianTotal": 3.7,
              "timeRange": "3.49 to 3.93",
              "medianOutput": 417.5,
              "outputRange": "392 to 456",
              "medianReasoning": 239.5,
              "reasoningRange": "214 to 278",
              "completedN": 4
            },
            {
              "config": "Claude Sonnet 5.5 (JSON schema) · Claude Code",
              "prompt": "Order to JSON",
              "strict": "4/4",
              "ci": "51% to 100%",
              "formatMisses": 0,
              "wrong": 0,
              "errors": 0,
              "fenced": 0,
              "medianTotal": 3.22,
              "timeRange": "2.95 to 6.14",
              "medianOutput": 273.5,
              "outputRange": "273 to 275",
              "medianReasoning": 64.5,
              "reasoningRange": "64 to 66",
              "completedN": 4
            },
            {
              "config": "Claude Sonnet 5.5 (JSON schema) · Claude Code",
              "prompt": "Invoice to JSON",
              "strict": "4/4",
              "ci": "51% to 100%",
              "formatMisses": 0,
              "wrong": 0,
              "errors": 0,
              "fenced": 0,
              "medianTotal": 4.37,
              "timeRange": "4.2 to 4.54",
              "medianOutput": 503.5,
              "outputRange": "495 to 541",
              "medianReasoning": 76,
              "reasoningRange": "0 to 154",
              "completedN": 4
            },
            {
              "config": "Claude Sonnet 5.5 (JSON schema) · Claude Code",
              "prompt": "Meeting notes to action items",
              "strict": "4/4",
              "ci": "51% to 100%",
              "formatMisses": 0,
              "wrong": 0,
              "errors": 0,
              "fenced": 0,
              "medianTotal": 4.25,
              "timeRange": "3.9 to 5.35",
              "medianOutput": 423.5,
              "outputRange": "408 to 498",
              "medianReasoning": 185.5,
              "reasoningRange": "170 to 260",
              "completedN": 4
            },
            {
              "config": "GPT-6.1 Sol (low, instructions) · Codex CLI",
              "prompt": "Order to JSON",
              "strict": "4/4",
              "ci": "51% to 100%",
              "formatMisses": 0,
              "wrong": 0,
              "errors": 0,
              "fenced": 0,
              "medianTotal": 6.21,
              "timeRange": "4.2 to 8.24",
              "medianOutput": 92,
              "outputRange": "69 to 93",
              "medianReasoning": 21,
              "reasoningRange": "0 to 22",
              "completedN": 4
            },
            {
              "config": "GPT-6.1 Sol (low, instructions) · Codex CLI",
              "prompt": "Invoice to JSON",
              "strict": "4/4",
              "ci": "51% to 100%",
              "formatMisses": 0,
              "wrong": 0,
              "errors": 0,
              "fenced": 0,
              "medianTotal": 10.19,
              "timeRange": "6.75 to 12.27",
              "medianOutput": 248,
              "outputRange": "235 to 259",
              "medianReasoning": 49,
              "reasoningRange": "36 to 60",
              "completedN": 4
            },
            {
              "config": "GPT-6.1 Sol (low, instructions) · Codex CLI",
              "prompt": "Meeting notes to action items",
              "strict": "4/4",
              "ci": "51% to 100%",
              "formatMisses": 0,
              "wrong": 0,
              "errors": 0,
              "fenced": 0,
              "medianTotal": 5.01,
              "timeRange": "4.71 to 5.81",
              "medianOutput": 117,
              "outputRange": "117 to 117",
              "medianReasoning": 0,
              "reasoningRange": "0 to 0",
              "completedN": 4
            },
            {
              "config": "GPT-6.1 Sol (low, JSON schema) · Codex CLI",
              "prompt": "Order to JSON",
              "strict": "4/4",
              "ci": "51% to 100%",
              "formatMisses": 0,
              "wrong": 0,
              "errors": 0,
              "fenced": 0,
              "medianTotal": 5.73,
              "timeRange": "4.62 to 20.97",
              "medianOutput": 99.5,
              "outputRange": "98 to 102",
              "medianReasoning": 22.5,
              "reasoningRange": "21 to 25",
              "completedN": 4
            },
            {
              "config": "GPT-6.1 Sol (low, JSON schema) · Codex CLI",
              "prompt": "Invoice to JSON",
              "strict": "4/4",
              "ci": "51% to 100%",
              "formatMisses": 0,
              "wrong": 0,
              "errors": 0,
              "fenced": 0,
              "medianTotal": 6.13,
              "timeRange": "5.77 to 6.37",
              "medianOutput": 172.5,
              "outputRange": "166 to 181",
              "medianReasoning": 40.5,
              "reasoningRange": "34 to 49",
              "completedN": 4
            },
            {
              "config": "GPT-6.1 Sol (low, JSON schema) · Codex CLI",
              "prompt": "Meeting notes to action items",
              "strict": "4/4",
              "ci": "51% to 100%",
              "formatMisses": 0,
              "wrong": 0,
              "errors": 0,
              "fenced": 0,
              "medianTotal": 5.76,
              "timeRange": "4.87 to 6.51",
              "medianOutput": 123,
              "outputRange": "123 to 123",
              "medianReasoning": 0,
              "reasoningRange": "0 to 0",
              "completedN": 4
            }
          ]
        },
        {
          "id": "structured-output-pairs",
          "title": "Paired view: the same prompt and repetition, asked both ways",
          "columns": [
            {
              "key": "model",
              "label": "Model and route",
              "unit": "text"
            },
            {
              "key": "pairs",
              "label": "Pairs",
              "unit": "count"
            },
            {
              "key": "both",
              "label": "Both passed",
              "unit": "count"
            },
            {
              "key": "onlyInstructions",
              "label": "Only instructions passed",
              "unit": "count"
            },
            {
              "key": "onlySchema",
              "label": "Only schema passed",
              "unit": "count"
            },
            {
              "key": "neither",
              "label": "Neither passed",
              "unit": "count"
            },
            {
              "key": "mcnemarP",
              "label": "Exact two-sided McNemar p (calculation)",
              "unit": "score"
            }
          ],
          "rows": [
            {
              "model": "Claude Haiku 4.5 · Claude Code",
              "pairs": 24,
              "both": 0,
              "onlyInstructions": 0,
              "onlySchema": 18,
              "neither": 6,
              "mcnemarP": 0.00000762939453125
            },
            {
              "model": "Claude Sonnet 5.5 · Claude Code",
              "pairs": 12,
              "both": 12,
              "onlyInstructions": 0,
              "onlySchema": 0,
              "neither": 0,
              "mcnemarP": 1
            },
            {
              "model": "GPT-6.1 Sol · Codex CLI",
              "pairs": 12,
              "both": 12,
              "onlyInstructions": 0,
              "onlySchema": 0,
              "neither": 0,
              "mcnemarP": 1
            }
          ]
        }
      ],
      "related": [
        "caching-consistency",
        "hard-model-head-to-head"
      ]
    },
    {
      "slug": "prompt-cache-across-sessions",
      "title": "Does a new Claude Code session reuse the prompt cache of an earlier one?",
      "seoTitle": "Claude Code prompt caching across sessions, tested",
      "description": "30 calls: later Claude Code sessions showed near-full turn-1 cache reads with a fixed folder, but not with new folders in this sample. Codex CLI tested too.",
      "question": "In the earlier caching study, a new Claude Code session did not read the cache that an earlier session wrote. Does a fixed working folder change that, and does putting the ledger in the system prompt help?",
      "answer": "Later Claude Code sessions showed near-full turn-1 cache reads with a fixed folder. Setup A (new folder): 0 of 2 met the 50% rule, with a 95% interval of 0% to 66%. Setup B (fixed folder): 2 of 2, with a 95% interval of 34% to 100%. Setup C (fixed folder): 2 of 2, with a 95% interval of 34% to 100%. The per-setup intervals overlap. Later A sessions read 1,463 tokens and wrote 6,386 to 6,388 tokens. Later B sessions read 7,833 tokens and wrote 0 tokens. Later C sessions read 7,815 tokens and wrote 0 tokens. We interpret A’s reads as a shared CLI prefix. No token-level trace tests this reading. A post-hoc calculation pools both studies. New folder: 0 of 6, with a 95% interval of 0% to 39%. Fixed folder: 4 of 4, with a 95% interval of 51% to 100%. These intervals do not overlap. The earlier sessions used another login, ledger and turn count, with interleaved models. This pool is a rough check, not a controlled comparison. Mean later-session turn-1 cost at list price was $0.0259 in A and $0.0016 across B and C (calculation). The ratio is 16.2 (calculation). The cost stats show n and ranges. These are subscription calls, not bills. Turn-1 times do not isolate a cache effect. Each setup has n = 3 sessions. A: median 1.56 s, range 1.34 s to 1.62 s. B: median 1.76 s, range 1.71 s to 1.86 s. C: median 1.77 s, range 1.69 s to 1.78 s. Ranges are not confidence intervals. Codex turn-1 cached input ranged from 0 to 8,960 tokens across 6 calls. Its highest read share was 56% (calculation). The 50% rule counts 2 of 4 later sessions, with a 95% interval of 15% to 85%. The post-hoc 90% rule counts 0 of 4, with a 95% interval of 0% to 49%. Neither rule proves which tokens came from the ledger. A and B differ in folder, ledger seed and call order. These observations do not show that a fixed folder is necessary or that the folder caused the difference. The surviving protocol file dates from after the calls. Treat this analysis as exploratory.",
      "date": "2026-10-07",
      "updated": "2026-10-07",
      "tags": [
        "prompt-caching",
        "cache-reuse",
        "claude-code",
        "codex-cli",
        "claude-sonnet",
        "gpt-6-1-sol",
        "working-folder",
        "calculation"
      ],
      "method": [
        "The protocol claims a declaration at 00:32 UTC. Its surviving file has a birth time of 00:49:27 UTC. The first counted call started at 00:32:33 UTC. We cannot verify a pre-call protocol.",
        "The run kept every try and made no retries. The run log lists three amendments. They cover the post-hoc Codex rule, a post-hoc pooled calculation and text corrections after a check.",
        "Claude Code 2.1.286 ran Sonnet 5.5 at its default effort. Tools were off. No MCP servers. No session persistence across processes. Caching was at the provider default.",
        "A session is one CLI process with 2 turns. Turn 1 asks a quantity lookup over a seeded synthetic stock ledger. The ledger has 100 lines and about 9,600 characters. Turn 2 asks a warehouse lookup. Each question has one reference answer. The checker trims spaces, quotes, backticks, a final period and currency units before comparison.",
        "Each setup had 3 sessions with 2 turns each (6 calls). A used a new folder each time. B used a fixed folder. Both put the ledger in the first user message. C used a fixed folder and put the ledger in the system prompt.",
        "Each setup has its own ledger seed. Different seeds prevent full ledger-prefix reuse between setups. Shared CLI-prefix reads remain possible. Sessions of a setup ran back to back. The gap between one session's last call and the next session's first call was 508 to 720 ms.",
        "A later session is session 2 or 3. The protocol labels it reuse when turn 1 read at least half of its input from the cache. This is a counter-based rule; no token-level trace identifies the ledger. The cache counters are the provider’s: uncached input, cache reads and cache writes (with the 5-minute and 1-hour split).",
        "List-price cost is a calculation. It prices uncached input at the input price and reads at the cache-read price. It prices 1-hour writes at 2× the input price. The calls ran on a subscription, so nothing here is a bill.",
        "Codex CLI ran GPT-6.1 Sol at medium effort in setups A and B. Each setup had 3 sessions with 2 turns each. Tools, apps, plugins and web search were off. Each session used an ephemeral read-only thread. The app-server reports input and cached input, but no cache writes. Its own large system prompt could explain a cached count.",
        "The surviving protocol records a 50% threshold but dates from after the calls. This rule counts 2 of 4 later Codex sessions. The post-hoc 90% rule counts 0 of 4. The respective Wilson 95% intervals are 15% to 85% and 0% to 49%. The highest turn-1 read share was 56% (calculation). No token-level trace identifies the ledger.",
        "One uncounted probe call per route, with its own ledger seed, checked the driver before the counted calls. The answer checker is the one of the earlier study. Its control test (every reference answer passes, every planted wrong answer fails) ran after the counted calls, because no cache measure depends on it.",
        "No batch stopped early. Nothing was trimmed. The Claude CLI reported a rate-limit status of \"allowed_warning\" on 9 of 18 Claude turns. No call was refused and the run did not stop.",
        "The normalized-answer checker passed 18 of 18 Claude answers (95% interval 82% to 100%).",
        "The normalized-answer checker passed 12 of 12 Codex answers (95% interval 76% to 100%)."
      ],
      "caveats": [
        "The surviving protocol file was created after all counted calls. Its claimed 00:32 UTC declaration is not supported by its file birth time. Amendment 1 and 2 state 00:36 and 00:37 UTC, but separate pre-edit copies do not verify those times. The current summary was regenerated at 07:41 UTC. Treat the analysis as exploratory.",
        "The synthetic ledger and short lookup questions reuse a designed task shape from the earlier study. All 30 answers passed after text normalization (95% interval 89% to 100%), so correctness hits a ceiling. This is a cache-counter probe, not evidence about coding quality or general task success.",
        "The protocol gives conflicting Codex entry gates: 30% weekly allowance remaining for the optional half, but 20% before the batch. The gate script enforces 20%. No retained gate receipt proves the allowance at run time. Call caps were kept: 18 Claude calls and 12 Codex calls, plus one probe each.",
        "The sample is small: 2 later sessions per setup and route. A count of 2 of 2 has a 95% interval of 34% to 100%, so the per-setup counts alone do not separate the setups. The token counts were stable: every later session in A read the same 1,463 tokens, and later sessions in B read 7,833 tokens each and those in C read 7,815 each. The pooled comparison across both studies is post hoc.",
        "The 4 earlier sessions in the pooled new-folder count differ from this study's sessions. They ran under a different Claude login, with another ledger and 5 turns per session. Sonnet and Opus sessions ran interleaved, so same-model sessions were 13 to 27 seconds apart (calculation from the recorded call times), not under 1 second. The pooled counts are a rough check, not a controlled comparison.",
        "Consecutive sessions of the same setup ran less than one second apart. The cache entries were 1-hour writes. This run says nothing about reuse after a longer gap or after an entry expires.",
        "A and B differ in working folder, ledger seed and call order. We did not test whether the folder path, its name or another property of a new empty folder breaks the match. The CLI documents an option that moves its per-machine system-prompt sections (the working folder is one) into the first user message. We did not test it.",
        "Setup C ran only with a fixed folder. We did not test whether a ledger in the system prompt protects the cache against a changing folder.",
        "Tools were off and the working folder was empty. A real repository adds other session-specific text (for example git state). We did not test that.",
        "Total turn-1 Claude input, including the ledger, is about 7.8k tokens. We interpret the 1,463 reads in later A sessions as the shared CLI prefix; this was not tested. We tested one CLI version and one model per route: Claude Code 2.1.286 with Sonnet 5.5, and Codex CLI 0.160.0 with GPT-6.1 Sol. Larger rewritten inputs cost more at these list prices (calculation); this does not predict another workload’s cache use.",
        "Codex CLI reports no cache-write count and has a large system prompt of its own. Its turn-1 cached counts varied from 0 to 8,960 in each setup. Neither setup reached the post-hoc 90% threshold. The small sample cannot show that folders never matter. We did not test why. Do not rank Codex against Claude on these numbers.",
        "The Codex criterion for a full read (90% or more of its input) was set after we saw the Codex counts (Amendment 1). Under the 50% threshold we recorded in the protocol, 2 of 4 later Codex sessions would count (8,960 tokens cached in each, the count the Codex probe call showed, which we read as the Codex system prefix). The 90% rule, set after we saw the counts, gives 0 of 4. Their Wilson 95% intervals are 15% to 85% and 0% to 49%, respectively.",
        "Costs are list-price calculations. The calls used flat subscriptions. Turn times come from a Mac that also ran other agent work, so contention can add noise."
      ],
      "sourceIds": [
        "agent-cache-sessions",
        "calc-cache-pricing",
        "price-anthropic",
        "agent-caching-consistency"
      ],
      "hero": {
        "statIds": [
          "cache-sessions-later-reuse-a",
          "cache-sessions-later-reuse-b"
        ]
      },
      "stats": [
        {
          "id": "cache-sessions-later-reuse-a",
          "label": "Later sessions with at least 50% of turn-1 input cached, A: new folder each time",
          "value": 0,
          "unit": "rate",
          "display": "0 of 2 (95% interval 0% to 66%)",
          "n": 2,
          "ci": [
            0,
            0.6576
          ],
          "note": "Claude Sonnet 5.5 in Claude Code. A later session is session 2 or 3. Reused = turn-1 read share of 0.5 or more. The surviving protocol records this rule but dates from after the calls. No token-level trace identifies the ledger."
        },
        {
          "id": "cache-sessions-later-reuse-b",
          "label": "Later sessions with at least 50% of turn-1 input cached, B: fixed folder",
          "value": 1,
          "unit": "rate",
          "display": "2 of 2 (95% interval 34% to 100%)",
          "n": 2,
          "ci": [
            0.3424,
            1
          ],
          "note": "Claude Sonnet 5.5 in Claude Code. A later session is session 2 or 3. Reused = turn-1 read share of 0.5 or more. The surviving protocol records this rule but dates from after the calls. No token-level trace identifies the ledger."
        },
        {
          "id": "cache-sessions-later-reuse-c",
          "label": "Later sessions with at least 50% of turn-1 input cached, C: fixed folder, ledger in system prompt",
          "value": 1,
          "unit": "rate",
          "display": "2 of 2 (95% interval 34% to 100%)",
          "n": 2,
          "ci": [
            0.3424,
            1
          ],
          "note": "Claude Sonnet 5.5 in Claude Code. A later session is session 2 or 3. Reused = turn-1 read share of 0.5 or more. The surviving protocol records this rule but dates from after the calls. No token-level trace identifies the ledger."
        },
        {
          "id": "cache-sessions-pooled-new-folder",
          "label": "Later sessions with at least 50% of turn-1 input cached, new folders, pooled across studies",
          "value": 0,
          "unit": "rate",
          "display": "0 of 6 (95% interval 0% to 39%)",
          "n": 6,
          "ci": [
            0,
            0.3903
          ],
          "note": "Calculation, post hoc (Amendment 2): this study’s setup A plus the earlier study’s 4 later Claude Code sessions (2 Sonnet, 2 Opus; 5 turns per session; another ledger). The 4 earlier sessions ran under a different Claude login, with Sonnet and Opus sessions interleaved: 13 to 27 seconds between a session’s last call and the next same-model session’s first call (calculation), against under 1 second here."
        },
        {
          "id": "cache-sessions-pooled-fixed-folder",
          "label": "Later sessions with at least 50% of turn-1 input cached, fixed folder, B and C pooled",
          "value": 1,
          "unit": "rate",
          "display": "4 of 4 (95% interval 51% to 100%)",
          "n": 4,
          "ci": [
            0.5101,
            1
          ],
          "note": "Calculation, post hoc (Amendment 2): setups B and C differ in where the ledger sits."
        },
        {
          "id": "cache-sessions-turn1-usd-new-folder",
          "label": "Turn-1 list-price cost of a later session, new folder each time (calculation)",
          "value": 0.025875,
          "unit": "usd",
          "display": "$0.0259",
          "n": 2,
          "note": "Mean of sessions 2 and 3, setup A; range $0.025871 to $0.025879. Calculation from recorded tokens and list prices; not a bill."
        },
        {
          "id": "cache-sessions-turn1-usd-fixed-folder",
          "label": "Turn-1 list-price cost of a later session, fixed folder (calculation)",
          "value": 0.001599,
          "unit": "usd",
          "display": "$0.0016",
          "n": 4,
          "note": "Mean of sessions 2 and 3 in setups B and C; range $0.001597 to $0.001601. Calculation from recorded tokens and list prices; not a bill."
        },
        {
          "id": "cache-sessions-turn1-cost-ratio",
          "label": "Turn-1 cost of a later session: new folder as a multiple of fixed folder (calculation)",
          "value": 16.18,
          "unit": "ratio",
          "display": "16.2×",
          "n": 6,
          "note": "Calculation. Turn-1 input here is about 7.8k tokens, including the ledger. A different workload can change both token use and cache matches."
        },
        {
          "id": "cache-sessions-codex-declared-threshold",
          "label": "Later Codex CLI sessions with at least 50% of turn-1 input cached (protocol threshold)",
          "value": 0.5,
          "unit": "rate",
          "display": "2 of 4 (95% interval 15% to 85%)",
          "n": 4,
          "ci": [
            0.15,
            0.85
          ],
          "note": "Counter threshold only. A prefix read can meet it; it does not prove that the ledger was read. The surviving protocol was created after the calls."
        },
        {
          "id": "cache-sessions-codex-later-reuse",
          "label": "Later Codex CLI sessions with at least 90% of turn-1 input cached (post-hoc rule) (setups A and B)",
          "value": 0,
          "unit": "rate",
          "display": "0 of 4 (95% interval 0% to 49%)",
          "n": 4,
          "ci": [
            0,
            0.4899
          ],
          "note": "GPT-6.1 Sol at medium effort. Reused = cached input of 90% or more of the input (Amendment 1, set after the counts were seen); the highest turn-1 cached share was 56%. Under the 50% threshold we recorded in the protocol, 2 of 4 later Codex sessions would count (8,960 tokens cached in each, the count the Codex probe call showed, which we read as the Codex system prefix). The 90% rule, set after we saw the counts, gives 0 of 4. Their Wilson 95% intervals are 15% to 85% and 0% to 49%, respectively."
        },
        {
          "id": "cache-sessions-calls",
          "label": "Counted calls in this study (every one counted)",
          "value": 30,
          "unit": "calls",
          "display": "30 (18 Claude Code, 12 Codex CLI), plus 2 uncounted probe calls"
        }
      ],
      "charts": [
        {
          "id": "cache-sessions-turn1-read-share",
          "title": "Claude Sonnet 5.5 · Claude Code: share of turn-1 input read from the cache, by setup and session",
          "subtitle": "One bar per call: turn 1 of one session; 3 sessions per setup, run back to back; read-share calculation",
          "kind": "grouped-bar",
          "unit": "rate",
          "polarity": "none",
          "xLabel": "Setup",
          "yLabel": "Input tokens read from cache",
          "series": [
            {
              "name": "Session 1 (first in its setup)",
              "points": [
                {
                  "label": "A: new folder each time",
                  "value": 0.0676,
                  "n": 1
                },
                {
                  "label": "B: fixed folder",
                  "value": 0.1867,
                  "n": 1
                },
                {
                  "label": "C: fixed folder, ledger in system prompt",
                  "value": 0.0691,
                  "n": 1
                }
              ]
            },
            {
              "name": "Session 2",
              "points": [
                {
                  "label": "A: new folder each time",
                  "value": 0.1863,
                  "n": 1
                },
                {
                  "label": "B: fixed folder",
                  "value": 0.9997,
                  "n": 1
                },
                {
                  "label": "C: fixed folder, ledger in system prompt",
                  "value": 0.9997,
                  "n": 1
                }
              ]
            },
            {
              "name": "Session 3",
              "points": [
                {
                  "label": "A: new folder each time",
                  "value": 0.1863,
                  "n": 1
                },
                {
                  "label": "B: fixed folder",
                  "value": 0.9997,
                  "n": 1
                },
                {
                  "label": "C: fixed folder, ledger in system prompt",
                  "value": 0.9997,
                  "n": 1
                }
              ]
            }
          ],
          "note": "Session 1 was the first session to use its setup’s ledger. Sessions 2 and 3 used that same ledger. Different seeds prevent full ledger-prefix reuse between setups; shared CLI-prefix reads remain possible. We interpret session-1 reads as a shared CLI prefix; no token-level trace proves this. Read share (calculation) = cache reads ÷ (uncached input + cache reads + cache writes), as the provider reports them. Each bar is one call, not a rate; the counts per setup are in the table. 2 later sessions per setup is a small number.",
          "sourceIds": [
            "agent-cache-sessions"
          ]
        },
        {
          "id": "cache-sessions-turn1-cost",
          "title": "Claude Sonnet 5.5 · Claude Code: list-price cost of turn 1, by setup and session (calculation)",
          "subtitle": "The recorded tokens of each call priced at list price; the last bar shows the mean of the 3 calls priced without any cache",
          "kind": "grouped-bar",
          "unit": "usd",
          "xLabel": "Setup",
          "yLabel": "USD (list price)",
          "series": [
            {
              "name": "Session 1 (first in its setup)",
              "points": [
                {
                  "label": "A: new folder each time",
                  "value": 0.02942,
                  "n": 1
                },
                {
                  "label": "B: fixed folder",
                  "value": 0.025807,
                  "n": 1
                },
                {
                  "label": "C: fixed folder, ledger in system prompt",
                  "value": 0.029242,
                  "n": 1
                }
              ]
            },
            {
              "name": "Session 2",
              "points": [
                {
                  "label": "A: new folder each time",
                  "value": 0.025871,
                  "n": 1
                },
                {
                  "label": "B: fixed folder",
                  "value": 0.001601,
                  "n": 1
                },
                {
                  "label": "C: fixed folder, ledger in system prompt",
                  "value": 0.001597,
                  "n": 1
                }
              ]
            },
            {
              "name": "Session 3",
              "points": [
                {
                  "label": "A: new folder each time",
                  "value": 0.025879,
                  "n": 1
                },
                {
                  "label": "B: fixed folder",
                  "value": 0.001601,
                  "n": 1
                },
                {
                  "label": "C: fixed folder, ledger in system prompt",
                  "value": 0.001597,
                  "n": 1
                }
              ]
            },
            {
              "name": "Same call without a cache (every input token at the input price)",
              "points": [
                {
                  "label": "A: new folder each time",
                  "value": 0.015735,
                  "n": 3
                },
                {
                  "label": "B: fixed folder",
                  "value": 0.0157,
                  "n": 3
                },
                {
                  "label": "C: fixed folder, ledger in system prompt",
                  "value": 0.015664,
                  "n": 3
                }
              ]
            }
          ],
          "note": "Calculation, not a bill: the calls ran on a subscription. Uncached input at the input price, cache reads at the cache-read price, 1-hour cache writes at 2× the input price (every write in this run was a 1-hour write); Sonnet 5.5 list prices, effective 2026-09-21. A write costs more than plain input, so a session that writes the ledger again costs more than no cache at all.",
          "sourceIds": [
            "agent-cache-sessions",
            "calc-cache-pricing",
            "price-anthropic"
          ]
        },
        {
          "id": "cache-sessions-turn-time",
          "title": "Claude Sonnet 5.5 · Claude Code: time per turn, by setup",
          "subtitle": "Median of 3 sessions; whiskers = fastest and slowest of the 3",
          "kind": "dot-range",
          "unit": "seconds",
          "yLabel": "Seconds",
          "series": [
            {
              "name": "Turn 1 (the ledger and question 1)",
              "points": [
                {
                  "label": "A: new folder each time",
                  "value": 1.56,
                  "lo": 1.34,
                  "hi": 1.62,
                  "n": 3
                },
                {
                  "label": "B: fixed folder",
                  "value": 1.76,
                  "lo": 1.71,
                  "hi": 1.86,
                  "n": 3
                },
                {
                  "label": "C: fixed folder, ledger in system prompt",
                  "value": 1.77,
                  "lo": 1.69,
                  "hi": 1.78,
                  "n": 3
                }
              ]
            },
            {
              "name": "Turn 2 (question 2)",
              "points": [
                {
                  "label": "A: new folder each time",
                  "value": 1.02,
                  "lo": 0.88,
                  "hi": 1.14,
                  "n": 3
                },
                {
                  "label": "B: fixed folder",
                  "value": 1.33,
                  "lo": 1.31,
                  "hi": 1.37,
                  "n": 3
                },
                {
                  "label": "C: fixed folder, ledger in system prompt",
                  "value": 1.13,
                  "lo": 1.09,
                  "hi": 1.29,
                  "n": 3
                }
              ]
            }
          ],
          "note": "Whiskers are a range (fastest and slowest of 3 sessions), not a confidence interval. Turn 2 read the cache in every setup. This design does not isolate a cache effect on speed. Ledger seed and call order also differ. The Mac also ran other agent work during these calls.",
          "whisker": "minmax",
          "sourceIds": [
            "agent-cache-sessions"
          ]
        },
        {
          "id": "cache-sessions-codex-turn1-cached",
          "title": "GPT-6.1 Sol (medium) · Codex CLI: cached input tokens on turn 1, by setup and session",
          "subtitle": "Turn-1 input was about 16,042 tokens in every call",
          "kind": "grouped-bar",
          "unit": "tokens",
          "xLabel": "Setup",
          "yLabel": "Cached input tokens",
          "series": [
            {
              "name": "Session 1 (first in its setup)",
              "points": [
                {
                  "label": "A: new folder each time",
                  "value": 4864,
                  "n": 1
                },
                {
                  "label": "B: fixed folder",
                  "value": 0,
                  "n": 1
                }
              ]
            },
            {
              "name": "Session 2",
              "points": [
                {
                  "label": "A: new folder each time",
                  "value": 8960,
                  "n": 1
                },
                {
                  "label": "B: fixed folder",
                  "value": 8960,
                  "n": 1
                }
              ]
            },
            {
              "name": "Session 3",
              "points": [
                {
                  "label": "A: new folder each time",
                  "value": 0,
                  "n": 1
                },
                {
                  "label": "B: fixed folder",
                  "value": 0,
                  "n": 1
                }
              ]
            }
          ],
          "note": "Cached input as the Codex app-server reports it (input includes the cached tokens; it reports no cache writes). Codex sends a large system prompt of its own, so a cached count of about 9,000 can come from that prefix alone. Each bar is one call. Not comparable with the Claude Code bars: different prefix, different cache.",
          "sourceIds": [
            "agent-cache-sessions"
          ]
        }
      ],
      "tables": [
        {
          "id": "cache-sessions-turns",
          "title": "Every counted turn",
          "columns": [
            {
              "key": "config",
              "label": "Route and model",
              "unit": "text"
            },
            {
              "key": "setup",
              "label": "Setup",
              "unit": "text"
            },
            {
              "key": "session",
              "label": "Session",
              "unit": "count"
            },
            {
              "key": "turn",
              "label": "Turn",
              "unit": "count"
            },
            {
              "key": "gap",
              "label": "Seconds since the previous session’s last call",
              "unit": "seconds"
            },
            {
              "key": "input",
              "label": "Input tokens (all)",
              "unit": "tokens"
            },
            {
              "key": "read",
              "label": "Read from cache",
              "unit": "tokens"
            },
            {
              "key": "written",
              "label": "Written to cache",
              "unit": "text"
            },
            {
              "key": "share",
              "label": "Read share",
              "unit": "rate"
            },
            {
              "key": "time",
              "label": "Time (s)",
              "unit": "seconds"
            },
            {
              "key": "withCache",
              "label": "USD with cache (calculation)",
              "unit": "usd"
            },
            {
              "key": "correct",
              "label": "Answer passes after normalization",
              "unit": "text"
            }
          ],
          "rows": [
            {
              "config": "Claude Sonnet 5.5 · Claude Code",
              "setup": "A: new folder each time",
              "session": 1,
              "turn": 1,
              "gap": null,
              "input": 7853,
              "read": 531,
              "written": 7320,
              "share": 0.0676,
              "time": 1.34,
              "withCache": 0.02942,
              "correct": "yes"
            },
            {
              "config": "Claude Sonnet 5.5 · Claude Code",
              "setup": "A: new folder each time",
              "session": 1,
              "turn": 2,
              "gap": null,
              "input": 7911,
              "read": 7851,
              "written": 58,
              "share": 0.9924,
              "time": 0.88,
              "withCache": 0.001856,
              "correct": "yes"
            },
            {
              "config": "Claude Sonnet 5.5 · Claude Code",
              "setup": "A: new folder each time",
              "session": 2,
              "turn": 1,
              "gap": 0.72,
              "input": 7851,
              "read": 1463,
              "written": 6386,
              "share": 0.1863,
              "time": 1.56,
              "withCache": 0.025871,
              "correct": "yes"
            },
            {
              "config": "Claude Sonnet 5.5 · Claude Code",
              "setup": "A: new folder each time",
              "session": 2,
              "turn": 2,
              "gap": null,
              "input": 7909,
              "read": 7849,
              "written": 58,
              "share": 0.9924,
              "time": 1.14,
              "withCache": 0.001856,
              "correct": "yes"
            },
            {
              "config": "Claude Sonnet 5.5 · Claude Code",
              "setup": "A: new folder each time",
              "session": 3,
              "turn": 1,
              "gap": 0.53,
              "input": 7853,
              "read": 1463,
              "written": 6388,
              "share": 0.1863,
              "time": 1.62,
              "withCache": 0.025879,
              "correct": "yes"
            },
            {
              "config": "Claude Sonnet 5.5 · Claude Code",
              "setup": "A: new folder each time",
              "session": 3,
              "turn": 2,
              "gap": null,
              "input": 7911,
              "read": 7851,
              "written": 58,
              "share": 0.9924,
              "time": 1.02,
              "withCache": 0.001856,
              "correct": "yes"
            },
            {
              "config": "Claude Sonnet 5.5 · Claude Code",
              "setup": "B: fixed folder",
              "session": 1,
              "turn": 1,
              "gap": null,
              "input": 7835,
              "read": 1463,
              "written": 6370,
              "share": 0.1867,
              "time": 1.86,
              "withCache": 0.025807,
              "correct": "yes"
            },
            {
              "config": "Claude Sonnet 5.5 · Claude Code",
              "setup": "B: fixed folder",
              "session": 1,
              "turn": 2,
              "gap": null,
              "input": 7893,
              "read": 7833,
              "written": 58,
              "share": 0.9924,
              "time": 1.33,
              "withCache": 0.001863,
              "correct": "yes"
            },
            {
              "config": "Claude Sonnet 5.5 · Claude Code",
              "setup": "B: fixed folder",
              "session": 2,
              "turn": 1,
              "gap": 0.51,
              "input": 7835,
              "read": 7833,
              "written": 0,
              "share": 0.9997,
              "time": 1.71,
              "withCache": 0.001601,
              "correct": "yes"
            },
            {
              "config": "Claude Sonnet 5.5 · Claude Code",
              "setup": "B: fixed folder",
              "session": 2,
              "turn": 2,
              "gap": null,
              "input": 7893,
              "read": 7891,
              "written": 0,
              "share": 0.9997,
              "time": 1.37,
              "withCache": 0.001642,
              "correct": "yes"
            },
            {
              "config": "Claude Sonnet 5.5 · Claude Code",
              "setup": "B: fixed folder",
              "session": 3,
              "turn": 1,
              "gap": 0.51,
              "input": 7835,
              "read": 7833,
              "written": 0,
              "share": 0.9997,
              "time": 1.76,
              "withCache": 0.001601,
              "correct": "yes"
            },
            {
              "config": "Claude Sonnet 5.5 · Claude Code",
              "setup": "B: fixed folder",
              "session": 3,
              "turn": 2,
              "gap": null,
              "input": 7893,
              "read": 7891,
              "written": 0,
              "share": 0.9997,
              "time": 1.31,
              "withCache": 0.001642,
              "correct": "yes"
            },
            {
              "config": "Claude Sonnet 5.5 · Claude Code",
              "setup": "C: fixed folder, ledger in system prompt",
              "session": 1,
              "turn": 1,
              "gap": null,
              "input": 7817,
              "read": 540,
              "written": 7275,
              "share": 0.0691,
              "time": 1.69,
              "withCache": 0.029242,
              "correct": "yes"
            },
            {
              "config": "Claude Sonnet 5.5 · Claude Code",
              "setup": "C: fixed folder, ledger in system prompt",
              "session": 1,
              "turn": 2,
              "gap": null,
              "input": 7875,
              "read": 7815,
              "written": 58,
              "share": 0.9924,
              "time": 1.29,
              "withCache": 0.001849,
              "correct": "yes"
            },
            {
              "config": "Claude Sonnet 5.5 · Claude Code",
              "setup": "C: fixed folder, ledger in system prompt",
              "session": 2,
              "turn": 1,
              "gap": 0.59,
              "input": 7817,
              "read": 7815,
              "written": 0,
              "share": 0.9997,
              "time": 1.78,
              "withCache": 0.001597,
              "correct": "yes"
            },
            {
              "config": "Claude Sonnet 5.5 · Claude Code",
              "setup": "C: fixed folder, ledger in system prompt",
              "session": 2,
              "turn": 2,
              "gap": null,
              "input": 7875,
              "read": 7873,
              "written": 0,
              "share": 0.9997,
              "time": 1.09,
              "withCache": 0.001629,
              "correct": "yes"
            },
            {
              "config": "Claude Sonnet 5.5 · Claude Code",
              "setup": "C: fixed folder, ledger in system prompt",
              "session": 3,
              "turn": 1,
              "gap": 0.69,
              "input": 7817,
              "read": 7815,
              "written": 0,
              "share": 0.9997,
              "time": 1.77,
              "withCache": 0.001597,
              "correct": "yes"
            },
            {
              "config": "Claude Sonnet 5.5 · Claude Code",
              "setup": "C: fixed folder, ledger in system prompt",
              "session": 3,
              "turn": 2,
              "gap": null,
              "input": 7875,
              "read": 7873,
              "written": 0,
              "share": 0.9997,
              "time": 1.13,
              "withCache": 0.001629,
              "correct": "yes"
            },
            {
              "config": "GPT-6.1 Sol (medium) · Codex CLI",
              "setup": "A: new folder each time",
              "session": 1,
              "turn": 1,
              "gap": null,
              "input": 16049,
              "read": 4864,
              "written": "not reported",
              "share": 0.3031,
              "time": 2.77,
              "withCache": null,
              "correct": "yes"
            },
            {
              "config": "GPT-6.1 Sol (medium) · Codex CLI",
              "setup": "A: new folder each time",
              "session": 1,
              "turn": 2,
              "gap": null,
              "input": 16078,
              "read": 15872,
              "written": "not reported",
              "share": 0.9872,
              "time": 1.55,
              "withCache": null,
              "correct": "yes"
            },
            {
              "config": "GPT-6.1 Sol (medium) · Codex CLI",
              "setup": "A: new folder each time",
              "session": 2,
              "turn": 1,
              "gap": 0.51,
              "input": 16051,
              "read": 8960,
              "written": "not reported",
              "share": 0.5582,
              "time": 3.98,
              "withCache": null,
              "correct": "yes"
            },
            {
              "config": "GPT-6.1 Sol (medium) · Codex CLI",
              "setup": "A: new folder each time",
              "session": 2,
              "turn": 2,
              "gap": null,
              "input": 16080,
              "read": 15872,
              "written": "not reported",
              "share": 0.9871,
              "time": 1.65,
              "withCache": null,
              "correct": "yes"
            },
            {
              "config": "GPT-6.1 Sol (medium) · Codex CLI",
              "setup": "A: new folder each time",
              "session": 3,
              "turn": 1,
              "gap": 0.58,
              "input": 16053,
              "read": 0,
              "written": "not reported",
              "share": 0,
              "time": 4.13,
              "withCache": null,
              "correct": "yes"
            },
            {
              "config": "GPT-6.1 Sol (medium) · Codex CLI",
              "setup": "A: new folder each time",
              "session": 3,
              "turn": 2,
              "gap": null,
              "input": 16082,
              "read": 15872,
              "written": "not reported",
              "share": 0.9869,
              "time": 1.68,
              "withCache": null,
              "correct": "yes"
            },
            {
              "config": "GPT-6.1 Sol (medium) · Codex CLI",
              "setup": "B: fixed folder",
              "session": 1,
              "turn": 1,
              "gap": null,
              "input": 16033,
              "read": 0,
              "written": "not reported",
              "share": 0,
              "time": 2.56,
              "withCache": null,
              "correct": "yes"
            },
            {
              "config": "GPT-6.1 Sol (medium) · Codex CLI",
              "setup": "B: fixed folder",
              "session": 1,
              "turn": 2,
              "gap": null,
              "input": 16062,
              "read": 15872,
              "written": "not reported",
              "share": 0.9882,
              "time": 1.54,
              "withCache": null,
              "correct": "yes"
            },
            {
              "config": "GPT-6.1 Sol (medium) · Codex CLI",
              "setup": "B: fixed folder",
              "session": 2,
              "turn": 1,
              "gap": 0.56,
              "input": 16033,
              "read": 8960,
              "written": "not reported",
              "share": 0.5588,
              "time": 2.82,
              "withCache": null,
              "correct": "yes"
            },
            {
              "config": "GPT-6.1 Sol (medium) · Codex CLI",
              "setup": "B: fixed folder",
              "session": 2,
              "turn": 2,
              "gap": null,
              "input": 16062,
              "read": 15872,
              "written": "not reported",
              "share": 0.9882,
              "time": 1.58,
              "withCache": null,
              "correct": "yes"
            },
            {
              "config": "GPT-6.1 Sol (medium) · Codex CLI",
              "setup": "B: fixed folder",
              "session": 3,
              "turn": 1,
              "gap": 0.63,
              "input": 16033,
              "read": 0,
              "written": "not reported",
              "share": 0,
              "time": 3.93,
              "withCache": null,
              "correct": "yes"
            },
            {
              "config": "GPT-6.1 Sol (medium) · Codex CLI",
              "setup": "B: fixed folder",
              "session": 3,
              "turn": 2,
              "gap": null,
              "input": 16062,
              "read": 15872,
              "written": "not reported",
              "share": 0.9882,
              "time": 2.9,
              "withCache": null,
              "correct": "yes"
            }
          ]
        }
      ],
      "related": [
        "caching-consistency"
      ]
    },
    {
      "slug": "prompt-cache-break-even",
      "title": "Prompt cache break-even: after how many reuses does a cached prefix cost less?",
      "seoTitle": "Prompt cache break-even by model and session length",
      "description": "A calculation on list prices: reuses before a cached prompt prefix costs less, per Claude and GPT model, and cost per 1,000 sessions of 1 to 20 turns.",
      "question": "With the cache-write surcharge Anthropic lists, after how many reuses does a cached prompt prefix cost less than no cache? What does that mean for a session of 1 to 20 turns, for each model’s price, and for a workload split across sessions?",
      "answer": "Calculation: a wholly new prefix with a 1-hour write costs less from the 3rd request, after 2 reuses. All four Claude price rows list a 1-hour write at twice the input price. Exact break-even is 1.03 to 1.11 reuses. With 19% already cached, it needs 1 reuse. This uses a pooled share from n = 6 sessions; session range 18.680% to 18.689%. A 5-minute write (an assumed 1.25 times the input price) needs 1 reuse. GPT-6.1 Sol and GPT-6 Luna list no write surcharge in the price list, so they save from the first reuse. They still do at the 1.25× write that another product table lists. For 1,000 10-turn sessions, the Sonnet 1-hour prefix cost is $45.42 against $156.62 without caching: 71.0% less. The prefix is a rounded mean of 7,831 tokens (n = 3; range 7,831 to 7,832). A one-turn session costs 2.0 times as much with the cache ($31.32 against $15.66 per 1,000). Each recorded session ran in a new temporary folder. Of 4 later sessions, 0 met the reuse proxy (95% Wilson 0% to 49%). The split calculation assumes ten full writes; it is not a replay of those partly cached sessions. The Opus 5.5 cache-read price is open: $0.2 or $0.4 per million. At $0.4, the break-even moves from 1.05 to 1.11 reuses, and the 10-turn saving from 75.5% to 71.0%.",
      "date": "2026-10-06",
      "updated": "2026-10-06",
      "tags": [
        "prompt-caching",
        "break-even",
        "thought-experiment",
        "calculation",
        "llm-pricing",
        "claude-haiku",
        "claude-sonnet",
        "claude-opus",
        "claude-fable",
        "gpt-6-1-sol",
        "gpt-6-luna"
      ],
      "method": [
        "This study makes no new model call and has no raw file of its own. It calculates from two inputs. The first is the turn tokens that the caching study recorded in 6 Claude Code sessions (3 on Sonnet 5.5, 3 on Opus 5.5). The second is the list prices in the price sources.",
        "A prefix of T tokens goes out in k + 1 requests. The first request writes it to the cache. The k later requests, the reuses, read it. Prices are USD per million tokens: input i, cache read r, cache write w.",
        "No cache: (k + 1) × T × i. A 1-hour write: T × w + k × T × r, where w is the listed 1-hour write price (twice i for every Anthropic model in the price list). A 5-minute write: T × 1.25 × i + k × T × r.",
        "The 1.25 is the multiplier that the caching study protocol states as Anthropic’s published figure. It is not in the price list. It is an assumption here. No 5-minute write occurred in the recorded sessions.",
        "Break-even: the cached prefix costs less when k > (w − i) ÷ (i − r). The exact break-even is that fraction. The reuses needed is the smallest non-negative whole number above it. A negative fraction means the first request already costs less, with zero reuses.",
        "A share f of the prefix can already be in the cache when the first request arrives. That request reads the share and writes the rest: T × (f × r + (1 − f) × w) + k × T × r.",
        "In the recorded sessions, turn 1 read 1,463 tokens on its first request. The usage counters do not record read/write timing. That is f = 19% (8,778 of 46,978 tokens, 6 sessions). Its origin was not isolated. The dollar figures use f = 0, the whole prefix new, which is the stricter case. The break-even rows show both.",
        "T is the rounded mean turn-1 input (uncached input + cache reads + cache writes) of a model’s recorded sessions. For Sonnet 5.5 it is 7,831 tokens (range 7,831 to 7,832, n = 3). For Opus 5.5 it is 7,828 tokens (the same in every session, n = 3). A session of N turns is N requests that send the prefix, so k = N − 1. The cost tables give 1,000 sessions of 1, 2, 3, 5, 10 and 20 turns.",
        "Split case: 10 requests as one 10-turn session (one write, 9 reads) against 10 one-turn sessions (10 writes). Each recorded session ran in a fresh process and a new temporary working folder. Of 4 later sessions, 0 met the reuse proxy. The split case assumes a wholly new prefix with no shared reuse; the receipts do not test that exact case. The reuse proxy follows the caching study’s rule: turn 1 reads more than it writes.",
        "Opus 5.5 price question: the price list gives a cache read of $0.2 per million (5% of the $4 input price). Another table of the product lists $0.4 (10%). This study did not check the vendor price. Every Opus 5.5 row and the second Opus line show both values.",
        "Check against the recorded sessions, on the input side only (output excluded). The recorded 5-turn sessions saved 55.4% on Sonnet 5.5 (n = 3; session range 54.3% to 57.1%) and 59.2% on Opus 5.5 (n = 3; range 58.5% to 59.7%). These ranges are not confidence intervals. The formula gives 52.0% and 56.0% with a new prefix, and 59.1% and 63.3% with 19% already cached. The two formula cases bracket the recorded figure for both models. The recorded sessions also wrote the tokens that each turn added.",
        "With output included, the recorded saving is 50.0% for Sonnet 5.5 (n = 3; session range 47.5% to 54.1%) and 53.1% for Opus 5.5 (n = 3; session range 51.4% to 54.3%). These are list-price calculations. The ranges are not confidence intervals.",
        "Everything is a list-price calculation, not a bill. The sample behind T is n = 3 sessions per model."
      ],
      "caveats": [
        "The retained Claude protocol file was created at 14:43:55 UTC, after the first counted session at 14:35:39 UTC on 2026-10-06. Its declaration says 14:25 UTC, but file times do not verify that claim. Treat this as a retrospective protocol record.",
        "The source has 30 attempted Claude turns, 0 failed turns and 0 turns without token usage. Failed turns with usage remain in cost totals. Missing usage cannot be priced. No quality rate or cache-caused speed effect is claimed.",
        "7 counted turns record a provider rate-limit warning status. The retained outcome log says there was no warning. The protocol calls for a stop on limit text in an error or warning. These receipts do not establish whether that text appeared; this discrepancy remains unresolved.",
        "One shared Mac and one synthetic ledger supplied these tokens. The ledger was shortened once after a two-call probe and before the counted sessions. All 30 counted answers passed, so this task set hits a quality ceiling. This calculation does not rank model quality or speed.",
        "A calculation, not a run and not a bill. The recorded calls used flat subscriptions. The prices are list prices dated 2026-09-21 (Anthropic) and 2026-10-03 (OpenAI); they can change.",
        "The formula prices the shared prefix only. A real turn also writes new tokens to the cache at the write price. In the recorded sessions that was 58 to 1,117 tokens per turn after turn 1. A real turn also produces output, which costs the same with or without the cache. The check against the recorded sessions shows the size of this effect.",
        "The formula assumes that every reuse arrives before the cache entry expires. An entry lasts 5 minutes or 1 hour, by write type. When the gap between requests is longer, the write repeats and the saving shrinks.",
        "The recorded sessions wrote 1-hour entries only (0 of 30 turns wrote a 5-minute entry). The 5-minute rows use an assumed 1.25 multiplier that this study did not test.",
        "Only 0 of 4 later sessions met the read-more-than-write proxy (95% Wilson 0% to 49%). They still read some cached tokens. This proxy does not establish cache provenance. Each recorded session ran in a fresh process and a new temporary working folder. The caching study protocol names that folder as one untested hypothesis for the miss, and this study did not test it. The split case assumes no shared reuse and a wholly new prefix. It does not price the partly cached recorded first turns. A script that keeps one working folder may read an earlier session’s cache: check your own receipts before you use the split case.",
        "The 19% already-cached share is not a constant. We recorded it on Sonnet 5.5 and Opus 5.5 sessions. For Haiku 4.5 and Fable 5.1 it is a what-if. The one Haiku 4.5 probe session (outside every cell) read 0 tokens from the cache on turn 1.",
        "The Opus 5.5 cache-read price is open: $0.2 or $0.4 per million. This study shows both. Resolve the price before you quote an Opus figure.",
        "The dollar figures depend on T: one synthetic ledger in one CLI version, 3 sessions per model. A different prefix changes the dollars. It does not change the break-even reuses, which depend on the price ratios only.",
        "The GPT rows follow the price list, which has no write surcharge. Another table of the product lists a write price of 1.25× input for both models. At that price they need 1 reuse (exact break-even 0.26 and 0.28), and a prefix used once costs 1.25 times as much. We did not check the vendor price. The Codex app-server reports no cache writes, and this study did not test reuse for GPT. The caching study measured Codex reads on a larger context."
      ],
      "sourceIds": [
        "calc-cache-pricing",
        "agent-caching-consistency",
        "price-anthropic",
        "price-openai"
      ],
      "stats": [
        {
          "id": "cache-break-even-1h-reuses",
          "label": "Reuses before a 1-hour cached prefix costs less, whole prefix new (calculation)",
          "value": 2,
          "unit": "count",
          "display": "2 reuses (the 3rd request)",
          "note": "Same for Haiku 4.5, Sonnet 5.5, Opus 5.5 and Fable 5.1. Exact break-even 1.03 to 1.11 reuses."
        },
        {
          "id": "cache-break-even-1h-reuses-recorded",
          "label": "Reuses before a 1-hour cached prefix costs less, 19% of it already cached as a pooled recorded share (calculation)",
          "value": 1,
          "unit": "count",
          "display": "1 reuse (the 2nd request)",
          "note": "Exact break-even 0.65 to 0.72 reuses. The recorded Claude Code sessions paid back on turn 2 (see the recorded-payback stat)."
        },
        {
          "id": "cache-break-even-5m-reuses",
          "label": "Reuses before a 5-minute cached prefix costs less, whole prefix new (calculation, assumed 1.25× write)",
          "value": 1,
          "unit": "count",
          "display": "1 reuse (the 2nd request)",
          "note": "Exact break-even 0.26 to 0.28 reuses. The 1.25 multiplier is an assumption."
        },
        {
          "id": "cache-break-even-no-surcharge-reuses",
          "label": "Reuses before a cached prefix costs less when the list price has no write surcharge (GPT-6.1 Sol and GPT-6 Luna; calculation)",
          "value": 1,
          "unit": "count",
          "display": "1 reuse (saves from the first read)",
          "note": "Exact break-even 0 reuses: a write is plain input, so the first read is already cheaper than sending the prefix again. Another table of the product lists a write price of 1.25× input for both models. At that price the exact break-even is 0.26 reuses (GPT-6.1 Sol) and 0.28 reuses (GPT-6 Luna), so 1 reuse still pays. We did not check the vendor price."
        },
        {
          "id": "cache-break-even-prefix-sonnet",
          "label": "Mean turn-1 input in the Sonnet 5.5 sessions (calculation)",
          "value": 7831,
          "unit": "tokens",
          "display": "7,831 tokens",
          "n": 3,
          "note": "Rounded mean calculation from 3 sessions (range 7,831 to 7,832): uncached input + cache reads + cache writes on turn 1."
        },
        {
          "id": "cache-break-even-prefix-opus",
          "label": "Mean turn-1 input in the Opus 5.5 sessions (calculation)",
          "value": 7828,
          "unit": "tokens",
          "display": "7,828 tokens",
          "n": 3,
          "note": "Rounded mean calculation from 3 sessions (the same in every session)."
        },
        {
          "id": "cache-break-even-precached-share",
          "label": "Share of turn-1 input read from the cache (calculation; origin not isolated)",
          "value": 0.1869,
          "unit": "rate",
          "display": "19% (8,778 of 46,978 turn-1 tokens)",
          "n": 6,
          "note": "Calculation across 6 sessions; session share range 18.680% to 18.689%. This is not a binomial pass rate. This part cost the read price on turn 1, not the write price."
        },
        {
          "id": "cache-break-even-recorded-payback",
          "label": "Turn at which a recorded Claude Code session’s total input cost with the cache first fell below its cost with no cache (calculation on recorded tokens)",
          "value": 2,
          "unit": "count",
          "display": "2 turns (6 of 6 sessions)",
          "n": 6,
          "note": "Input side only, 1-hour writes at the list price, output left out. After turn 1 the cache had cost more, as the formula says for a prefix used once."
        },
        {
          "id": "cache-break-even-once-penalty",
          "label": "Cost of caching a prefix that is used once, as a multiple of no cache (1-hour write; calculation)",
          "value": 2,
          "unit": "ratio",
          "display": "2.0x",
          "note": "A 5-minute write: 1.3x (assumption). Same for every Anthropic model in the price list."
        },
        {
          "id": "cache-break-even-sonnet-10-turns",
          "label": "Saving from a 1-hour cache over 10 turns, Sonnet 5.5 prefix (calculation)",
          "value": 0.71,
          "unit": "rate",
          "display": "71.0% ($45.42 vs $156.62 per 1,000 sessions)",
          "note": "Prefix of 7,831 tokens; whole prefix new; output left out."
        },
        {
          "id": "cache-break-even-opus-10-turns",
          "label": "Saving from a 1-hour cache over 10 turns, Opus 5.5 prefix, cache read $0.2 per M (calculation)",
          "value": 0.755,
          "unit": "rate",
          "display": "75.5% ($76.71 vs $313.12 per 1,000 sessions)",
          "note": "Prefix of 7,828 tokens; whole prefix new; output left out."
        },
        {
          "id": "cache-break-even-opus-read-price",
          "label": "Saving from a 1-hour cache over 10 turns, Opus 5.5 prefix, cache read $0.4 per M (calculation)",
          "value": 0.71,
          "unit": "rate",
          "display": "71.0% ($90.80 vs $313.12 per 1,000 sessions)",
          "note": "Break-even 1.11 reuses at $0.4 per M against 1.05 at $0.2 per M. The vendor price was not checked."
        },
        {
          "id": "cache-break-even-split-sonnet",
          "label": "10 one-turn sessions with no shared reuse with a 1-hour cache, as a multiple of one 10-turn session, Sonnet 5.5 prefix (calculation)",
          "value": 6.9,
          "unit": "ratio",
          "display": "6.9x ($313.24 vs $45.42 per 1,000 workloads)",
          "note": "With no cache the same 10 requests cost $156.62. Each recorded session ran in a new temporary folder, and 0 of 4 met the read-more-than-write proxy. The split case assumes a wholly new prefix with no shared reuse."
        },
        {
          "id": "cache-break-even-cross-session",
          "label": "Later sessions whose first turn read more tokens than it wrote (reuse proxy)",
          "value": 0,
          "unit": "count",
          "display": "0 of 4",
          "n": 4,
          "note": "95% Wilson 0% to 49%, n = 4 later sessions. Same proxy as the caching study: reads exceed writes. It cannot identify cache provenance. The cause was not tested."
        },
        {
          "id": "cache-break-even-5m-writes-recorded",
          "label": "Recorded Claude turns that wrote a 5-minute cache entry",
          "value": 0,
          "unit": "count",
          "display": "0 of 30",
          "n": 30,
          "note": "Every recorded write was a 1-hour write, so the 1.25 multiplier of the 5-minute rows was not tested."
        },
        {
          "id": "cache-break-even-check-sonnet",
          "label": "Saving over 5 turns on the input side, recorded vs the formula (calculation), Sonnet 5.5",
          "value": 0.5535,
          "unit": "rate",
          "display": "55.4% recorded (formula 52.0% with a new prefix, 59.1% with 19% already cached)",
          "n": 3,
          "note": "Calculation on recorded tokens and list prices; session saving range 54.3% to 57.1%, not a confidence interval. The recorded figure includes new turn tokens; the formula does not."
        },
        {
          "id": "cache-break-even-check-opus",
          "label": "Saving over 5 turns on the input side, recorded vs the formula (calculation), Opus 5.5",
          "value": 0.5918,
          "unit": "rate",
          "display": "59.2% recorded (formula 56.0% with a new prefix, 63.3% with 19% already cached)",
          "n": 3,
          "note": "Calculation on recorded tokens and list prices; session saving range 58.5% to 59.7%, not a confidence interval. Opus 5.5 cache read at $0.2 per M."
        }
      ],
      "charts": [
        {
          "id": "cache-break-even-reads",
          "title": "Reuses before a cached prefix costs less, by model and write type (calculation)",
          "subtitle": "The break-even point: reuses at which the cached and the uncached cost are equal, at list prices",
          "kind": "grouped-bar",
          "unit": "score",
          "yLabel": "Reuses at break-even",
          "series": [
            {
              "name": "1-hour write (2× input), whole prefix new",
              "points": [
                {
                  "label": "Claude Haiku 4.5",
                  "value": 1.11
                },
                {
                  "label": "Claude Sonnet 5.5",
                  "value": 1.11
                },
                {
                  "label": "Claude Opus 5.5 (cache read $0.2 per M)",
                  "value": 1.05
                },
                {
                  "label": "Claude Opus 5.5 (cache read $0.4 per M)",
                  "value": 1.11
                },
                {
                  "label": "Claude Fable 5.1",
                  "value": 1.03
                }
              ]
            },
            {
              "name": "1-hour write, pooled n = 6 session share, 19% already cached (as recorded)",
              "points": [
                {
                  "label": "Claude Haiku 4.5",
                  "value": 0.72
                },
                {
                  "label": "Claude Sonnet 5.5",
                  "value": 0.72
                },
                {
                  "label": "Claude Opus 5.5 (cache read $0.2 per M)",
                  "value": 0.67
                },
                {
                  "label": "Claude Opus 5.5 (cache read $0.4 per M)",
                  "value": 0.72
                },
                {
                  "label": "Claude Fable 5.1",
                  "value": 0.65
                }
              ]
            },
            {
              "name": "5-minute write (1.25× input, an assumption)",
              "points": [
                {
                  "label": "Claude Haiku 4.5",
                  "value": 0.28
                },
                {
                  "label": "Claude Sonnet 5.5",
                  "value": 0.28
                },
                {
                  "label": "Claude Opus 5.5 (cache read $0.2 per M)",
                  "value": 0.26
                },
                {
                  "label": "Claude Opus 5.5 (cache read $0.4 per M)",
                  "value": 0.28
                },
                {
                  "label": "Claude Fable 5.1",
                  "value": 0.26
                }
              ]
            }
          ],
          "note": "Calculation, not a run: break-even reuses = (write price − input price) ÷ (input price − cache-read price). A cached prefix costs less once the reuses pass that point. On a new prefix, a 1-hour write needs 2 reuses (the 3rd request). With 19% already cached it needs 1 reuse, and a 5-minute write needs 1 reuse. The 1.25 of the 5-minute write is an assumption. The caching study protocol states it as Anthropic’s published figure. No 5-minute write occurred in the recorded sessions. We recorded the 19% on Sonnet 5.5 and Opus 5.5 sessions. For Haiku 4.5 and Fable 5.1 it is a what-if. GPT-6.1 Sol and GPT-6 Luna list no write surcharge, so their break-even is 0 reuses and the chart leaves them out. The price list gives Opus 5.5 a cache read of $0.2 per million. Another table of the product lists $0.4. We did not check the vendor price, so the chart shows both.",
          "sourceIds": [
            "calc-cache-pricing",
            "price-anthropic",
            "price-openai"
          ]
        },
        {
          "id": "cache-break-even-cost-curve",
          "title": "Cost of a reused prefix with and without the cache, by session length (calculation)",
          "subtitle": "Prefix of 7,831 tokens (Sonnet 5.5) and 7,828 tokens (Opus 5.5), the rounded mean turn-1 input; n = 3 Sonnet sessions (range 7,831 to 7,832) and 3 Opus sessions (the same in every session); USD per 1,000 sessions",
          "kind": "line",
          "unit": "usd",
          "xLabel": "Turns in the session (requests that send the prefix)",
          "yLabel": "USD per 1,000 sessions (list price)",
          "series": [
            {
              "name": "Claude Sonnet 5.5 (no cache)",
              "points": [
                {
                  "label": "1 turn",
                  "value": 15.66
                },
                {
                  "label": "2 turns",
                  "value": 31.32
                },
                {
                  "label": "3 turns",
                  "value": 46.99
                },
                {
                  "label": "5 turns",
                  "value": 78.31
                },
                {
                  "label": "10 turns",
                  "value": 156.62
                },
                {
                  "label": "20 turns",
                  "value": 313.24
                }
              ]
            },
            {
              "name": "Claude Sonnet 5.5 (1-hour cache write)",
              "points": [
                {
                  "label": "1 turn",
                  "value": 31.32
                },
                {
                  "label": "2 turns",
                  "value": 32.89
                },
                {
                  "label": "3 turns",
                  "value": 34.46
                },
                {
                  "label": "5 turns",
                  "value": 37.59
                },
                {
                  "label": "10 turns",
                  "value": 45.42
                },
                {
                  "label": "20 turns",
                  "value": 61.08
                }
              ]
            },
            {
              "name": "Claude Opus 5.5 (no cache)",
              "points": [
                {
                  "label": "1 turn",
                  "value": 31.31
                },
                {
                  "label": "2 turns",
                  "value": 62.62
                },
                {
                  "label": "3 turns",
                  "value": 93.94
                },
                {
                  "label": "5 turns",
                  "value": 156.56
                },
                {
                  "label": "10 turns",
                  "value": 313.12
                },
                {
                  "label": "20 turns",
                  "value": 626.24
                }
              ]
            },
            {
              "name": "Claude Opus 5.5 (1-hour cache write, read $0.2 per M)",
              "points": [
                {
                  "label": "1 turn",
                  "value": 62.62
                },
                {
                  "label": "2 turns",
                  "value": 64.19
                },
                {
                  "label": "3 turns",
                  "value": 65.76
                },
                {
                  "label": "5 turns",
                  "value": 68.89
                },
                {
                  "label": "10 turns",
                  "value": 76.71
                },
                {
                  "label": "20 turns",
                  "value": 92.37
                }
              ]
            },
            {
              "name": "Claude Opus 5.5 (1-hour cache write, read $0.4 per M)",
              "points": [
                {
                  "label": "1 turn",
                  "value": 62.62
                },
                {
                  "label": "2 turns",
                  "value": 65.76
                },
                {
                  "label": "3 turns",
                  "value": 68.89
                },
                {
                  "label": "5 turns",
                  "value": 75.15
                },
                {
                  "label": "10 turns",
                  "value": 90.8
                },
                {
                  "label": "20 turns",
                  "value": 122.12
                }
              ]
            }
          ],
          "note": "Calculation, not a bill. The chart multiplies list prices by a prefix of 7,831 tokens (Sonnet) or 7,828 tokens (Opus). It uses a 1-hour write at 2× input and treats the whole prefix as new. It leaves out output and the tokens that each turn adds. The cache costs more at 1 to 2 turns and less from 3. The second Opus line uses a $0.4 cache read, which another table of the product lists. We did not check the vendor price. The table adds the 5-minute write.",
          "sourceIds": [
            "calc-cache-pricing",
            "agent-caching-consistency",
            "price-anthropic",
            "price-openai"
          ]
        },
        {
          "id": "cache-break-even-session-split",
          "title": "One 10-turn session or ten 1-turn sessions: cost with and without the cache (calculation)",
          "subtitle": "USD per 1,000 workloads of 10 requests that send the same prefix; each recorded session started in a new temporary folder, n = 4 later sessions; 0 met the read-more-than-write proxy, 95% Wilson 0% to 49%",
          "kind": "grouped-bar",
          "unit": "usd",
          "yLabel": "USD per 1,000 workloads of 10 requests (list price)",
          "series": [
            {
              "name": "No cache (the same either way)",
              "points": [
                {
                  "label": "Claude Sonnet 5.5",
                  "value": 156.62
                },
                {
                  "label": "Claude Opus 5.5 (cache read $0.2 per M)",
                  "value": 313.12
                },
                {
                  "label": "Claude Opus 5.5 (cache read $0.4 per M)",
                  "value": 313.12
                }
              ]
            },
            {
              "name": "One 10-turn session, 1-hour cache",
              "points": [
                {
                  "label": "Claude Sonnet 5.5",
                  "value": 45.42
                },
                {
                  "label": "Claude Opus 5.5 (cache read $0.2 per M)",
                  "value": 76.71
                },
                {
                  "label": "Claude Opus 5.5 (cache read $0.4 per M)",
                  "value": 90.8
                }
              ]
            },
            {
              "name": "Ten 1-turn sessions, 1-hour cache (assumed no reuse of the new prefix)",
              "points": [
                {
                  "label": "Claude Sonnet 5.5",
                  "value": 313.24
                },
                {
                  "label": "Claude Opus 5.5 (cache read $0.2 per M)",
                  "value": 626.24
                },
                {
                  "label": "Claude Opus 5.5 (cache read $0.4 per M)",
                  "value": 626.24
                }
              ]
            }
          ],
          "note": "Calculation, not a bill: the same prefix, prices and 1-hour write as the cost-curve chart. Each recorded session ran in a new temporary working folder. On turn 1, 0 of 4 later sessions read more tokens than they wrote. This proxy cannot identify which earlier call supplied a cache entry. The split calculation assumes no shared reuse and a wholly new prefix, so it charges ten full writes. With reuse across sessions they would cost the same as one 10-turn session. The cause of the missing reuse was not tested here. The caching study protocol names the random temporary folder as one untested hypothesis. The table adds the 5-minute variants.",
          "sourceIds": [
            "calc-cache-pricing",
            "agent-caching-consistency",
            "price-anthropic",
            "price-openai"
          ]
        }
      ],
      "tables": [
        {
          "id": "cache-break-even-models",
          "title": "Break-even reuses by model (calculation)",
          "columns": [
            {
              "key": "model",
              "label": "Priced as",
              "unit": "text"
            },
            {
              "key": "input",
              "label": "Input $/M",
              "unit": "usd"
            },
            {
              "key": "read",
              "label": "Cache read $/M",
              "unit": "usd"
            },
            {
              "key": "write1h",
              "label": "1-hour write $/M",
              "unit": "usd"
            },
            {
              "key": "write5m",
              "label": "5-minute write $/M (assumed 1.25×)",
              "unit": "usd"
            },
            {
              "key": "be1h",
              "label": "Break-even, 1-hour, new prefix",
              "unit": "score"
            },
            {
              "key": "whole1h",
              "label": "Reuses needed, 1-hour, new prefix",
              "unit": "count"
            },
            {
              "key": "be1hRec",
              "label": "Break-even, 1-hour, 19% already cached",
              "unit": "score"
            },
            {
              "key": "whole1hRec",
              "label": "Reuses needed, 1-hour, 19% already cached",
              "unit": "count"
            },
            {
              "key": "be5m",
              "label": "Break-even, 5-minute",
              "unit": "score"
            },
            {
              "key": "whole5m",
              "label": "Reuses needed, 5-minute",
              "unit": "count"
            },
            {
              "key": "note",
              "label": "Note",
              "unit": "text"
            }
          ],
          "rows": [
            {
              "model": "Claude Haiku 4.5",
              "input": 1,
              "read": 0.1,
              "write1h": 2,
              "write5m": 1.25,
              "be1h": 1.11,
              "whole1h": 2,
              "be1hRec": 0.72,
              "whole1hRec": 1,
              "be5m": 0.28,
              "whole5m": 1,
              "note": "Anthropic list price, 1-hour write at 2× input"
            },
            {
              "model": "Claude Sonnet 5.5",
              "input": 2,
              "read": 0.2,
              "write1h": 4,
              "write5m": 2.5,
              "be1h": 1.11,
              "whole1h": 2,
              "be1hRec": 0.72,
              "whole1hRec": 1,
              "be5m": 0.28,
              "whole5m": 1,
              "note": "Anthropic list price, 1-hour write at 2× input"
            },
            {
              "model": "Claude Opus 5.5 (cache read $0.2 per M)",
              "input": 4,
              "read": 0.2,
              "write1h": 8,
              "write5m": 5,
              "be1h": 1.05,
              "whole1h": 2,
              "be1hRec": 0.67,
              "whole1hRec": 1,
              "be5m": 0.26,
              "whole5m": 1,
              "note": "Anthropic list price, 1-hour write at 2× input"
            },
            {
              "model": "Claude Opus 5.5 (cache read $0.4 per M)",
              "input": 4,
              "read": 0.4,
              "write1h": 8,
              "write5m": 5,
              "be1h": 1.11,
              "whole1h": 2,
              "be1hRec": 0.72,
              "whole1hRec": 1,
              "be5m": 0.28,
              "whole5m": 1,
              "note": "Cache-read price from another product table; the vendor price was not checked"
            },
            {
              "model": "Claude Fable 5.1",
              "input": 10,
              "read": 0.25,
              "write1h": 20,
              "write5m": 12.5,
              "be1h": 1.03,
              "whole1h": 2,
              "be1hRec": 0.65,
              "whole1hRec": 1,
              "be5m": 0.26,
              "whole5m": 1,
              "note": "Anthropic list price, 1-hour write at 2× input"
            },
            {
              "model": "GPT-6.1 Sol",
              "input": 2,
              "read": 0.1,
              "write1h": 2,
              "write5m": 2,
              "be1h": 0,
              "whole1h": 1,
              "be1hRec": -0.19,
              "whole1hRec": 0,
              "be5m": 0,
              "whole5m": 1,
              "note": "No write surcharge in the price list: a write is plain input, so the cache saves from the first read. Another product table lists 1.25× input as the write price: break-even 0.26 reuses"
            },
            {
              "model": "GPT-6 Luna",
              "input": 0.1,
              "read": 0.01,
              "write1h": 0.1,
              "write5m": 0.1,
              "be1h": 0,
              "whole1h": 1,
              "be1hRec": -0.19,
              "whole1hRec": 0,
              "be5m": 0,
              "whole5m": 1,
              "note": "No write surcharge in the price list: a write is plain input, so the cache saves from the first read. Another product table lists 1.25× input as the write price: break-even 0.28 reuses"
            }
          ]
        },
        {
          "id": "cache-break-even-saving-by-turns",
          "title": "Saving from the cache by session length (calculation; negative = the cache costs more)",
          "columns": [
            {
              "key": "model",
              "label": "Priced as",
              "unit": "text"
            },
            {
              "key": "write",
              "label": "Write",
              "unit": "text"
            },
            {
              "key": "t1",
              "label": "1 turn",
              "unit": "rate"
            },
            {
              "key": "t2",
              "label": "2 turns",
              "unit": "rate"
            },
            {
              "key": "t3",
              "label": "3 turns",
              "unit": "rate"
            },
            {
              "key": "t5",
              "label": "5 turns",
              "unit": "rate"
            },
            {
              "key": "t10",
              "label": "10 turns",
              "unit": "rate"
            },
            {
              "key": "t20",
              "label": "20 turns",
              "unit": "rate"
            }
          ],
          "rows": [
            {
              "model": "Claude Haiku 4.5",
              "write": "1-hour write, new prefix",
              "t1": -1,
              "t2": -0.05,
              "t3": 0.2667,
              "t5": 0.52,
              "t10": 0.71,
              "t20": 0.805
            },
            {
              "model": "Claude Haiku 4.5",
              "write": "5-minute write (assumed 1.25×)",
              "t1": -0.25,
              "t2": 0.325,
              "t3": 0.5167,
              "t5": 0.67,
              "t10": 0.785,
              "t20": 0.8425
            },
            {
              "model": "Claude Sonnet 5.5",
              "write": "1-hour write, new prefix",
              "t1": -1,
              "t2": -0.05,
              "t3": 0.2667,
              "t5": 0.52,
              "t10": 0.71,
              "t20": 0.805
            },
            {
              "model": "Claude Sonnet 5.5",
              "write": "5-minute write (assumed 1.25×)",
              "t1": -0.25,
              "t2": 0.325,
              "t3": 0.5167,
              "t5": 0.67,
              "t10": 0.785,
              "t20": 0.8425
            },
            {
              "model": "Claude Opus 5.5 (cache read $0.2 per M)",
              "write": "1-hour write, new prefix",
              "t1": -1,
              "t2": -0.025,
              "t3": 0.3,
              "t5": 0.56,
              "t10": 0.755,
              "t20": 0.8525
            },
            {
              "model": "Claude Opus 5.5 (cache read $0.2 per M)",
              "write": "5-minute write (assumed 1.25×)",
              "t1": -0.25,
              "t2": 0.35,
              "t3": 0.55,
              "t5": 0.71,
              "t10": 0.83,
              "t20": 0.89
            },
            {
              "model": "Claude Opus 5.5 (cache read $0.4 per M)",
              "write": "1-hour write, new prefix",
              "t1": -1,
              "t2": -0.05,
              "t3": 0.2667,
              "t5": 0.52,
              "t10": 0.71,
              "t20": 0.805
            },
            {
              "model": "Claude Opus 5.5 (cache read $0.4 per M)",
              "write": "5-minute write (assumed 1.25×)",
              "t1": -0.25,
              "t2": 0.325,
              "t3": 0.5167,
              "t5": 0.67,
              "t10": 0.785,
              "t20": 0.8425
            },
            {
              "model": "Claude Fable 5.1",
              "write": "1-hour write, new prefix",
              "t1": -1,
              "t2": -0.0125,
              "t3": 0.3167,
              "t5": 0.58,
              "t10": 0.7775,
              "t20": 0.8763
            },
            {
              "model": "Claude Fable 5.1",
              "write": "5-minute write (assumed 1.25×)",
              "t1": -0.25,
              "t2": 0.3625,
              "t3": 0.5667,
              "t5": 0.73,
              "t10": 0.8525,
              "t20": 0.9138
            },
            {
              "model": "GPT-6.1 Sol",
              "write": "no write surcharge",
              "t1": 0,
              "t2": 0.475,
              "t3": 0.6333,
              "t5": 0.76,
              "t10": 0.855,
              "t20": 0.9025
            },
            {
              "model": "GPT-6 Luna",
              "write": "no write surcharge",
              "t1": 0,
              "t2": 0.45,
              "t3": 0.6,
              "t5": 0.72,
              "t10": 0.81,
              "t20": 0.855
            }
          ]
        },
        {
          "id": "cache-break-even-cost-per-1000",
          "title": "Cost per 1,000 sessions by session length (calculation)",
          "columns": [
            {
              "key": "line",
              "label": "Priced as",
              "unit": "text"
            },
            {
              "key": "prefix",
              "label": "Prefix (tokens)",
              "unit": "tokens"
            },
            {
              "key": "turns",
              "label": "Turns",
              "unit": "count"
            },
            {
              "key": "none",
              "label": "No cache (USD)",
              "unit": "usd"
            },
            {
              "key": "c1h",
              "label": "1-hour write (USD)",
              "unit": "usd"
            },
            {
              "key": "c5m",
              "label": "5-minute write, assumed 1.25× (USD)",
              "unit": "usd"
            },
            {
              "key": "save1h",
              "label": "Saving, 1-hour",
              "unit": "rate"
            },
            {
              "key": "save5m",
              "label": "Saving, 5-minute",
              "unit": "rate"
            }
          ],
          "rows": [
            {
              "line": "Claude Sonnet 5.5",
              "prefix": 7831,
              "turns": 1,
              "none": 15.66,
              "c1h": 31.32,
              "c5m": 19.58,
              "save1h": -1,
              "save5m": -0.25
            },
            {
              "line": "Claude Sonnet 5.5",
              "prefix": 7831,
              "turns": 2,
              "none": 31.32,
              "c1h": 32.89,
              "c5m": 21.14,
              "save1h": -0.05,
              "save5m": 0.325
            },
            {
              "line": "Claude Sonnet 5.5",
              "prefix": 7831,
              "turns": 3,
              "none": 46.99,
              "c1h": 34.46,
              "c5m": 22.71,
              "save1h": 0.2667,
              "save5m": 0.5167
            },
            {
              "line": "Claude Sonnet 5.5",
              "prefix": 7831,
              "turns": 5,
              "none": 78.31,
              "c1h": 37.59,
              "c5m": 25.84,
              "save1h": 0.52,
              "save5m": 0.67
            },
            {
              "line": "Claude Sonnet 5.5",
              "prefix": 7831,
              "turns": 10,
              "none": 156.62,
              "c1h": 45.42,
              "c5m": 33.67,
              "save1h": 0.71,
              "save5m": 0.785
            },
            {
              "line": "Claude Sonnet 5.5",
              "prefix": 7831,
              "turns": 20,
              "none": 313.24,
              "c1h": 61.08,
              "c5m": 49.34,
              "save1h": 0.805,
              "save5m": 0.8425
            },
            {
              "line": "Claude Opus 5.5 (cache read $0.2 per M)",
              "prefix": 7828,
              "turns": 1,
              "none": 31.31,
              "c1h": 62.62,
              "c5m": 39.14,
              "save1h": -1,
              "save5m": -0.25
            },
            {
              "line": "Claude Opus 5.5 (cache read $0.2 per M)",
              "prefix": 7828,
              "turns": 2,
              "none": 62.62,
              "c1h": 64.19,
              "c5m": 40.71,
              "save1h": -0.025,
              "save5m": 0.35
            },
            {
              "line": "Claude Opus 5.5 (cache read $0.2 per M)",
              "prefix": 7828,
              "turns": 3,
              "none": 93.94,
              "c1h": 65.76,
              "c5m": 42.27,
              "save1h": 0.3,
              "save5m": 0.55
            },
            {
              "line": "Claude Opus 5.5 (cache read $0.2 per M)",
              "prefix": 7828,
              "turns": 5,
              "none": 156.56,
              "c1h": 68.89,
              "c5m": 45.4,
              "save1h": 0.56,
              "save5m": 0.71
            },
            {
              "line": "Claude Opus 5.5 (cache read $0.2 per M)",
              "prefix": 7828,
              "turns": 10,
              "none": 313.12,
              "c1h": 76.71,
              "c5m": 53.23,
              "save1h": 0.755,
              "save5m": 0.83
            },
            {
              "line": "Claude Opus 5.5 (cache read $0.2 per M)",
              "prefix": 7828,
              "turns": 20,
              "none": 626.24,
              "c1h": 92.37,
              "c5m": 68.89,
              "save1h": 0.8525,
              "save5m": 0.89
            },
            {
              "line": "Claude Opus 5.5 (cache read $0.4 per M)",
              "prefix": 7828,
              "turns": 1,
              "none": 31.31,
              "c1h": 62.62,
              "c5m": 39.14,
              "save1h": -1,
              "save5m": -0.25
            },
            {
              "line": "Claude Opus 5.5 (cache read $0.4 per M)",
              "prefix": 7828,
              "turns": 2,
              "none": 62.62,
              "c1h": 65.76,
              "c5m": 42.27,
              "save1h": -0.05,
              "save5m": 0.325
            },
            {
              "line": "Claude Opus 5.5 (cache read $0.4 per M)",
              "prefix": 7828,
              "turns": 3,
              "none": 93.94,
              "c1h": 68.89,
              "c5m": 45.4,
              "save1h": 0.2667,
              "save5m": 0.5167
            },
            {
              "line": "Claude Opus 5.5 (cache read $0.4 per M)",
              "prefix": 7828,
              "turns": 5,
              "none": 156.56,
              "c1h": 75.15,
              "c5m": 51.66,
              "save1h": 0.52,
              "save5m": 0.67
            },
            {
              "line": "Claude Opus 5.5 (cache read $0.4 per M)",
              "prefix": 7828,
              "turns": 10,
              "none": 313.12,
              "c1h": 90.8,
              "c5m": 67.32,
              "save1h": 0.71,
              "save5m": 0.785
            },
            {
              "line": "Claude Opus 5.5 (cache read $0.4 per M)",
              "prefix": 7828,
              "turns": 20,
              "none": 626.24,
              "c1h": 122.12,
              "c5m": 98.63,
              "save1h": 0.805,
              "save5m": 0.8425
            }
          ]
        },
        {
          "id": "cache-break-even-split",
          "title": "10 requests as one session or as 10 one-turn sessions with no shared reuse, per 1,000 workloads (calculation)",
          "columns": [
            {
              "key": "line",
              "label": "Priced as",
              "unit": "text"
            },
            {
              "key": "none",
              "label": "No cache (USD)",
              "unit": "usd"
            },
            {
              "key": "one1h",
              "label": "One 10-turn session, 1-hour (USD)",
              "unit": "usd"
            },
            {
              "key": "ten1h",
              "label": "10 one-turn sessions with no shared reuse, 1-hour (USD)",
              "unit": "usd"
            },
            {
              "key": "one5m",
              "label": "One 10-turn session, 5-minute (USD)",
              "unit": "usd"
            },
            {
              "key": "ten5m",
              "label": "10 one-turn sessions with no shared reuse, 5-minute (USD)",
              "unit": "usd"
            },
            {
              "key": "ratio",
              "label": "10 one-turn sessions with no shared reuse vs one session, 1-hour",
              "unit": "ratio"
            }
          ],
          "rows": [
            {
              "line": "Claude Sonnet 5.5",
              "none": 156.62,
              "one1h": 45.42,
              "ten1h": 313.24,
              "one5m": 33.67,
              "ten5m": 195.78,
              "ratio": 6.9
            },
            {
              "line": "Claude Opus 5.5 (cache read $0.2 per M)",
              "none": 313.12,
              "one1h": 76.71,
              "ten1h": 626.24,
              "one5m": 53.23,
              "ten5m": 391.4,
              "ratio": 8.2
            },
            {
              "line": "Claude Opus 5.5 (cache read $0.4 per M)",
              "none": 313.12,
              "one1h": 90.8,
              "ten1h": 626.24,
              "one5m": 67.32,
              "ten5m": 391.4,
              "ratio": 6.9
            }
          ]
        },
        {
          "id": "cache-break-even-inputs",
          "title": "Recorded inputs and a check of the formula (calculation)",
          "columns": [
            {
              "key": "model",
              "label": "Recorded sessions",
              "unit": "text"
            },
            {
              "key": "sessions",
              "label": "Sessions",
              "unit": "count"
            },
            {
              "key": "prefix",
              "label": "Turn-1 input, rounded mean (tokens)",
              "unit": "tokens"
            },
            {
              "key": "range",
              "label": "Turn-1 prefix, range (tokens)",
              "unit": "text"
            },
            {
              "key": "already",
              "label": "Already cached at turn 1, mean (tokens)",
              "unit": "tokens"
            },
            {
              "key": "written",
              "label": "Written on turn 1, mean (tokens)",
              "unit": "tokens"
            },
            {
              "key": "savingRange",
              "label": "Input saving, session range (not a confidence interval)",
              "unit": "text"
            },
            {
              "key": "recordedInput",
              "label": "Recorded saving, input side, 5 turns",
              "unit": "rate"
            },
            {
              "key": "newPrefix",
              "label": "Formula, whole prefix new, 5 turns",
              "unit": "rate"
            },
            {
              "key": "precached",
              "label": "Formula, 19% already cached, 5 turns",
              "unit": "rate"
            },
            {
              "key": "wholeSavingRange",
              "label": "Saving with output, session range (not a confidence interval)",
              "unit": "text"
            },
            {
              "key": "recordedWhole",
              "label": "Recorded saving, with output, 5 turns",
              "unit": "rate"
            }
          ],
          "rows": [
            {
              "model": "Claude Sonnet 5.5 · Claude Code",
              "sessions": 3,
              "prefix": 7831,
              "range": "7,831 to 7,832",
              "already": 1463,
              "written": 6366,
              "savingRange": "54.3% to 57.1%",
              "wholeSavingRange": "47.5% to 54.1%",
              "recordedInput": 0.5535,
              "newPrefix": 0.52,
              "precached": 0.591,
              "recordedWhole": 0.4996
            },
            {
              "model": "Claude Opus 5.5 · Claude Code",
              "sessions": 3,
              "prefix": 7828,
              "range": "7,828",
              "already": 1463,
              "written": 6363,
              "savingRange": "58.5% to 59.7%",
              "wholeSavingRange": "51.4% to 54.3%",
              "recordedInput": 0.5918,
              "newPrefix": 0.56,
              "precached": 0.6329,
              "recordedWhole": 0.5313
            }
          ]
        }
      ],
      "related": [
        "caching-consistency",
        "cost-thought-experiments"
      ]
    },
    {
      "slug": "routing-holdout",
      "title": "Jev vs Claude routers on unseen decisions: a blind holdout",
      "seoTitle": "Jev vs Claude router accuracy on unseen decisions",
      "description": "Jev 1.13, Claude Haiku 4.5 and Sonnet 5.5 on 56 routing decisions written blind and frozen first: exact rate with 95% intervals, tuned vs unseen.",
      "question": "Does a router keep its accuracy on typed routing decisions that nobody tuned against its answers?",
      "answer": "The holdout does not establish a router ranking. The author wrote 56 new decisions blind to router answers. Labels froze before the first router call. Jev 1.13 (TypeSafe): 46 of 56 (82%, 95% interval 70% to 90%). Claude Haiku 4.5 · Claude Code: 44 of 56 (79%, 95% interval 66% to 87%). Claude Sonnet 5.5 (low) · Claude Code: 49 of 56 (88%, 95% interval 76% to 94%). The intervals overlap, so the holdout does not rank the routers. The exact McNemar tests find no significant difference (p from 0.18 to 0.688). These paired tests have no multiple-test correction. Tuned set → holdout exact rate (calculation; counts and intervals in the chart): Jev 1.13 90% → 82%, Haiku 4.5 89% → 79%, Sonnet 5.5 (low) 94% → 88%. All 3 observed rates were lower. Each router’s tuned and holdout intervals overlap. The data does not show a clear drop for any router. Decision groups (calculation; counts and intervals in the table). On the 3 one-question types the observed exact rate changed by Jev 1.13 −9.5, Haiku 4.5 −5.1, Sonnet 5.5 (low) −2.4 points. These are calculations, not a ranking of drops. On context shape the observed exact rate changed by Jev 1.13 −17.9, Haiku 4.5 −39.3, Sonnet 5.5 (low) −27.2 points. These are calculations, not a ranking of drops. Every tuned and holdout pair of intervals overlaps. The data neither shows nor rules out a home advantage for Jev. The 3 drops range from 6.4 to 10.5 points (calculation). The sets differ in mix: context shape is 32 of 82 tuned cases and 14 of 56 holdout cases. Secondary scoring keeps only questions where our set accepts the second label. Our set accepts 115 of 125 (92%, 95% interval 86% to 96%). Our first label agrees on 100 of 125 (80%, 95% interval 72% to 86%). Jev 1.13: 46 of 56 (82%, 95% interval 70% to 90%). Haiku 4.5: 46 of 56 (82%, 95% interval 70% to 90%). Sonnet 5.5 (low): 52 of 56 (93%, 95% interval 83% to 97%). The weakest label is “artifacts”. Our set accepts the second label on 1 of 9 (11%, 95% interval 2% to 43%). Set-aware kappa is 0 (calculation). This label broke our protocol. Matches to our label: Jev 1.13: 9 of 9 (100%, 95% interval 70% to 100%). Haiku 4.5: 4 of 9 (44%, 95% interval 19% to 73%). Sonnet 5.5 (low): 7 of 9 (78%, 95% interval 45% to 94%). See the caveats. Exact by decision type, of 14 each (95% intervals in the chart): Failure class 13 to 14; Message intent 12 to 14; Is it a rule? 13; Context shape 5 to 8. Failure class, Message intent and Is it a rule? are near the ceiling for every router (85% or more), so most differences come from context shape. Cost per 1,000 decisions (calculation): Jev 1.13 $0.0307, Haiku 4.5 $7.13, Sonnet 5.5 (low) $7.24. Median time per decision: Jev 1.13: 139 ms (p50–p95 139 ms to 192 ms, n = 168). Haiku 4.5: 9.44 s (p50–p95 9.44 s to 25.41 s, n = 56). Sonnet 5.5 (low): 2.36 s (p50–p95 2.36 s to 3.66 s, n = 56). These ranges are not confidence intervals. Jev used direct HTTPS; Claude used its CLI. The shared Mac also ran local arena model servers from 15:18 local time (20:18 UTC), before all holdout calls. Host contention may affect wall times, especially CLI times. We did not rerun the timing.",
      "date": "2026-10-06",
      "updated": "2026-10-06",
      "tags": [
        "routing",
        "jev",
        "claude-haiku",
        "claude-sonnet",
        "model-routing",
        "holdout",
        "benchmark-method"
      ],
      "method": [
        "Cases: 56 new cases, 14 per decision type, in the production case shape. The types are failure class, message intent, is-it-a-rule and context shape. Production asks a router about each case. We score only questions the rule leaves open: 125 labelled questions.",
        "Blindness: the author says they opened no router answer before the label freeze. File times support this order but cannot prove blindness. A script compared every case with existing cases. The author rewrote 17 close cases before the freeze.",
        "File times: the protocol file dates from 21:35:45 UTC on 2026-10-06. Case drafts and one uncounted second-labeller probe came first. The first counted labelling call came at 21:35:49 UTC. The freeze came at 21:37:02 UTC. The frozen file and its checksum preceded every router call.",
        "Second labeller: GPT-6.1 Sol through the Codex CLI, 4 counted calls. It labelled every case without the author labels.\n\nOur set accepts its label on 115 of 125 (92%, 95% interval 86% to 96%). It matches our first label on 100 of 125 (80%, 95% interval 72% to 86%). Of these questions, 24 accept two or three labels. The lowest agreement was on “artifacts”; see the caveats. Secondary scoring keeps only questions where our set accepts its label.",
        "Routers: Jev 1.13 used direct HTTPS with the production request body. It made 168 calls in 3 repetitions, with no retries.\n\nClaude Haiku 4.5 used CLI default effort; Claude Sonnet 5.5 used effort low. Both used the Claude Code CLI with the production system text and schema. Each made 56 counted calls, one per decision. Each route had one uncounted probe.\n\nTotal calls stayed within the caps: Jev 169/200, Claude 114/120, labeller 5/5. No arm stopped early or lost cases.",
        "Scoring: the repository’s decision-eval runner. Exact means every scored question in a case was acceptable. Key accuracy counts each question. We use Wilson 95% intervals and exact McNemar tests on paired cases. The paired tests have no multiple-test correction. Errors count as wrong.",
        "Cost: reported tokens × list price per 1,000 decisions, a calculation."
      ],
      "caveats": [
        "The case author and two of the three routers are Claude models, so a same-family label bias is possible. A second labeller (GPT-6.1 Sol) labelled every case blind; the secondary scoring keeps only the keys where its label is inside our acceptable set.",
        "56 cases (14 per decision type) from one author: per-type intervals are very wide.",
        "The routes differ: Jev ran over direct HTTPS, the Claude routers through the Claude Code CLI, which adds start-up time and tokens.",
        "Jev figures are its first repetition; repetitions 2 and 3 and the stability across all three are reported beside it.",
        "The artifacts label violates the protocol rule to leave judgement calls unscored. The author accepted only no on all 9 labelled cases. The second labeller chose no on 1 of 9. Frozen labels stay unchanged; sensitivity calculations are not new runs.",
        "The protocol file was created at 21:35:45.935 UTC on 2026-10-06. It followed the case drafts and an uncounted labeller probe. It preceded the first counted labelling call at 21:35:49.3 UTC, the label freeze at 21:37:02.582 UTC and every router call. It was not written before every call.",
        "The shared Mac also ran local arena model servers from 15:18 local time (20:18 UTC), before all holdout calls. Host contention may affect wall times, especially CLI times. We did not rerun the timing.",
        "The three one-question case sets are near a ceiling: every router scored 12 to 14 of 14 on each. The 95% Wilson interval for 14 of 14 is 78.5% to 100%. These sets provide little room to separate routers.",
        "Per-question Wilson intervals treat questions as independent. Several questions share each context-shape case, so those intervals can understate uncertainty. Label-agreement intervals have the same limit.",
        "Set-aware kappa uses the second label as the author label when the acceptable set contains it. This adaptive calculation can inflate agreement. Only kappaStrict compares two fixed labels.",
        "The tuned set was revised against Jev answers. Its size and decision mix differ from the holdout. Overlapping intervals neither show nor rule out a home advantage.",
        "The three paired McNemar tests have no multiple-test correction. Their p values do not prove equal accuracy.",
        "Failure class, Message intent and Is it a rule?: every router answered at least 85% of the 14 cases exactly. These case sets are near the ceiling and barely separate the routers; context shape carries most of the differences.",
        "Agreement with the second labeller counts a label inside our acceptable set: 115 of 125 questions (92%). 24 of the 125 questions accept two or three labels, so agreement with our first label alone is lower: 100 of 125 (80%).",
        "The “artifacts” label broke our own rules. The second labeller’s answer was inside our set on 1 of 9 cases (kappa 0). We labelled “no” on every one of them. The protocol leaves out a question whose answer is a judgement call, and the production case suites accept both answers for a fresh work item. Cases where the router’s answer equalled our label: Jev 1.13 9 of 9, Haiku 4.5 4 of 9, Sonnet 5.5 (low) 7 of 9. The scoring against our labels therefore favours Jev on this question. The labels stayed frozen.\n\nWith both answers accepted for “artifacts” and the same answers scored again (a calculation), exact answers are Jev 1.13 46 of 56 (82%, 95% interval 70% to 90%); Haiku 4.5 45 of 56 (80%, 95% interval 68% to 89%); Sonnet 5.5 (low) 50 of 56 (89%, 95% interval 79% to 95%). Context shape: Jev 1.13 8 of 14 (57%, 95% interval 33% to 79%); Haiku 4.5 6 of 14 (43%, 95% interval 21% to 67%); Sonnet 5.5 (low) 9 of 14 (64%, 95% interval 39% to 84%); the overall intervals still overlap.",
        "Label spread is narrow on some questions. One label holds at least 90% of the cases for artifacts (“no”, 9 of 9) and memories (“yes”, 9 of 10). One option has a single case for knowledge (“none”), scope (“large”) and transcript (“summary”), so those options are barely tested.",
        "The tuned and holdout sets differ in size and mix, so a change between them cannot be assigned to the router or to the case set. The data neither shows nor rules out a home advantage."
      ],
      "sourceIds": [
        "agent-routing-holdout",
        "agent-routing",
        "calc-repricing",
        "price-jev",
        "price-anthropic"
      ],
      "stats": [
        {
          "id": "holdout-exact-jev",
          "label": "Jev 1.13 (TypeSafe): exact on unseen decisions",
          "value": 0.8214,
          "unit": "rate",
          "display": "82% (46/56)",
          "n": 56,
          "ci": [
            0.7016,
            0.9
          ]
        },
        {
          "id": "holdout-exact-claude-haiku",
          "label": "Claude Haiku 4.5 · Claude Code: exact on unseen decisions",
          "value": 0.7857,
          "unit": "rate",
          "display": "79% (44/56)",
          "n": 56,
          "ci": [
            0.6618,
            0.8729
          ]
        },
        {
          "id": "holdout-exact-claude-sonnet",
          "label": "Claude Sonnet 5.5 (low) · Claude Code: exact on unseen decisions",
          "value": 0.875,
          "unit": "rate",
          "display": "88% (49/56)",
          "n": 56,
          "ci": [
            0.7637,
            0.9381
          ]
        },
        {
          "id": "holdout-jev-stability",
          "label": "Jev 1.13 (TypeSafe): same answers on every key in 3 repetitions",
          "value": 0.9464,
          "unit": "rate",
          "display": "95% (53/56)",
          "n": 56,
          "ci": [
            0.8539,
            0.9816
          ]
        },
        {
          "id": "holdout-jev-reps-range",
          "label": "Jev 1.13 (TypeSafe): exact in each of 3 repetitions",
          "value": 0.8214,
          "unit": "rate",
          "display": "46, 46, 47 of 56",
          "n": 56,
          "note": "Range 46 to 47 of 56; not a confidence interval. The value is the first repetition."
        },
        {
          "id": "holdout-label-agreement",
          "label": "Second labeller (GPT-6.1 Sol): inside our acceptable label sets",
          "value": 0.92,
          "unit": "rate",
          "display": "92% (115/125)",
          "n": 125,
          "ci": [
            0.859,
            0.956
          ],
          "note": "Per labelled question, across all four decision types. 24 of the 125 questions accept two or three labels, so this counts a second label that is inside our set, not only our first label. First label only: 100 of 125 (80%)."
        },
        {
          "id": "holdout-label-agreement-first",
          "label": "Second labeller (GPT-6.1 Sol): equal to our first label",
          "value": 0.8,
          "unit": "rate",
          "display": "80% (100/125)",
          "n": 125,
          "ci": [
            0.7214,
            0.8607
          ],
          "note": "Per labelled question. Our first label is the least inclusive one in our set. The inside-our-set figure is holdout-label-agreement."
        },
        {
          "id": "holdout-gap-jev",
          "label": "Jev 1.13 (TypeSafe): holdout minus tuned-set exact rate",
          "value": -0.081,
          "unit": "rate",
          "display": "−8.1 points",
          "n": 56,
          "note": "Calculation. Tuned 90% (82% to 95%, n = 82); holdout 82% (70% to 90%, n = 56). The intervals overlap."
        },
        {
          "id": "holdout-gap-claude-haiku",
          "label": "Claude Haiku 4.5 · Claude Code: holdout minus tuned-set exact rate",
          "value": -0.1045,
          "unit": "rate",
          "display": "−10.5 points",
          "n": 56,
          "note": "Calculation. Tuned 89% (80% to 94%, n = 82); holdout 79% (66% to 87%, n = 56). The intervals overlap."
        },
        {
          "id": "holdout-gap-claude-sonnet",
          "label": "Claude Sonnet 5.5 (low) · Claude Code: holdout minus tuned-set exact rate",
          "value": -0.064,
          "unit": "rate",
          "display": "−6.4 points",
          "n": 56,
          "note": "Calculation. Tuned 94% (87% to 97%, n = 82); holdout 88% (76% to 94%, n = 56). The intervals overlap."
        },
        {
          "id": "holdout-jev-cost-per-1000",
          "label": "Jev 1.13 (TypeSafe): cost per 1,000 unseen decisions",
          "value": 0.03065,
          "unit": "usd",
          "display": "$0.0307",
          "n": 168,
          "note": "Calculation: reported input tokens × $0.042 per million."
        }
      ],
      "charts": [
        {
          "id": "routing-holdout-exact",
          "title": "Unseen routing decisions answered exactly right",
          "subtitle": "Share of the 56 holdout cases where every scored question was acceptable",
          "kind": "dot-range",
          "unit": "rate",
          "polarity": "higher",
          "yLabel": "Exact",
          "series": [
            {
              "name": "Exact rate",
              "points": [
                {
                  "label": "Jev 1.13 (TypeSafe)",
                  "value": 0.8214,
                  "lo": 0.7016,
                  "hi": 0.9,
                  "n": 56,
                  "highlight": true
                },
                {
                  "label": "Claude Haiku 4.5 · Claude Code",
                  "value": 0.7857,
                  "lo": 0.6618,
                  "hi": 0.8729,
                  "n": 56,
                  "highlight": false
                },
                {
                  "label": "Claude Sonnet 5.5 (low) · Claude Code",
                  "value": 0.875,
                  "lo": 0.7637,
                  "hi": 0.9381,
                  "n": 56,
                  "highlight": false
                }
              ]
            }
          ],
          "note": "Whiskers are 95% Wilson intervals. Cases written blind to every router answer, then frozen. Jev: its first of three repetitions.",
          "whisker": "ci95",
          "sourceIds": [
            "agent-routing-holdout"
          ]
        },
        {
          "id": "routing-holdout-key-accuracy",
          "title": "Per-question accuracy on unseen decisions",
          "subtitle": "Each open question a router was asked; an unanswered question counts as wrong",
          "kind": "dot-range",
          "unit": "rate",
          "polarity": "higher",
          "yLabel": "Correct answers",
          "series": [
            {
              "name": "Key accuracy",
              "points": [
                {
                  "label": "Jev 1.13 (TypeSafe)",
                  "value": 0.904,
                  "lo": 0.8397,
                  "hi": 0.9442,
                  "n": 125,
                  "highlight": true
                },
                {
                  "label": "Claude Haiku 4.5 · Claude Code",
                  "value": 0.816,
                  "lo": 0.739,
                  "hi": 0.8741,
                  "n": 125,
                  "highlight": false
                },
                {
                  "label": "Claude Sonnet 5.5 (low) · Claude Code",
                  "value": 0.92,
                  "lo": 0.859,
                  "hi": 0.956,
                  "n": 125,
                  "highlight": false
                }
              ]
            }
          ],
          "note": "Whiskers are nominal 95% Wilson intervals. Context-shape cases ask up to 8 questions each, the other decision types one. Questions in one case are not independent; these intervals do not adjust for that grouping.",
          "whisker": "ci95",
          "sourceIds": [
            "agent-routing-holdout"
          ]
        },
        {
          "id": "routing-holdout-by-purpose",
          "title": "Exact rate on unseen decisions, by decision type",
          "subtitle": "14 cases per decision type",
          "kind": "grouped-bar",
          "unit": "rate",
          "polarity": "higher",
          "yLabel": "Exact",
          "series": [
            {
              "name": "Jev 1.13 (TypeSafe)",
              "points": [
                {
                  "label": "Failure class",
                  "value": 0.9286,
                  "lo": 0.6853,
                  "hi": 0.9873,
                  "n": 14,
                  "highlight": true
                },
                {
                  "label": "Message intent",
                  "value": 0.8571,
                  "lo": 0.6006,
                  "hi": 0.9599,
                  "n": 14,
                  "highlight": true
                },
                {
                  "label": "Is it a rule?",
                  "value": 0.9286,
                  "lo": 0.6853,
                  "hi": 0.9873,
                  "n": 14,
                  "highlight": true
                },
                {
                  "label": "Context shape",
                  "value": 0.5714,
                  "lo": 0.3259,
                  "hi": 0.7862,
                  "n": 14,
                  "highlight": true
                }
              ]
            },
            {
              "name": "Claude Haiku 4.5 · Claude Code",
              "points": [
                {
                  "label": "Failure class",
                  "value": 0.9286,
                  "lo": 0.6853,
                  "hi": 0.9873,
                  "n": 14,
                  "highlight": false
                },
                {
                  "label": "Message intent",
                  "value": 0.9286,
                  "lo": 0.6853,
                  "hi": 0.9873,
                  "n": 14,
                  "highlight": false
                },
                {
                  "label": "Is it a rule?",
                  "value": 0.9286,
                  "lo": 0.6853,
                  "hi": 0.9873,
                  "n": 14,
                  "highlight": false
                },
                {
                  "label": "Context shape",
                  "value": 0.3571,
                  "lo": 0.1634,
                  "hi": 0.6124,
                  "n": 14,
                  "highlight": false
                }
              ]
            },
            {
              "name": "Claude Sonnet 5.5 (low) · Claude Code",
              "points": [
                {
                  "label": "Failure class",
                  "value": 1,
                  "lo": 0.7847,
                  "hi": 1,
                  "n": 14,
                  "highlight": false
                },
                {
                  "label": "Message intent",
                  "value": 1,
                  "lo": 0.7847,
                  "hi": 1,
                  "n": 14,
                  "highlight": false
                },
                {
                  "label": "Is it a rule?",
                  "value": 0.9286,
                  "lo": 0.6853,
                  "hi": 0.9873,
                  "n": 14,
                  "highlight": false
                },
                {
                  "label": "Context shape",
                  "value": 0.5714,
                  "lo": 0.3259,
                  "hi": 0.7862,
                  "n": 14,
                  "highlight": false
                }
              ]
            }
          ],
          "note": "Whiskers are 95% Wilson intervals. With 14 cases a perfect score has an interval of 78% to 100%, so a decision type where every router scores 14 of 14 is at its ceiling and cannot rank them.",
          "whisker": "ci95",
          "sourceIds": [
            "agent-routing-holdout"
          ]
        },
        {
          "id": "routing-holdout-tuned-vs-unseen",
          "title": "Tuned case set vs unseen holdout: exact rate per router",
          "subtitle": "Tuned set: the routing study’s 82 cases, revised against Jev answers. Holdout: 56 new cases, frozen before any router call",
          "kind": "grouped-bar",
          "unit": "rate",
          "polarity": "higher",
          "yLabel": "Exact",
          "series": [
            {
              "name": "Tuned set (routing-jev-vs-llm)",
              "points": [
                {
                  "label": "Jev 1.13 (TypeSafe)",
                  "value": 0.9024,
                  "lo": 0.8191,
                  "hi": 0.9497,
                  "n": 82,
                  "highlight": true
                },
                {
                  "label": "Claude Haiku 4.5 · Claude Code",
                  "value": 0.8902,
                  "lo": 0.8044,
                  "hi": 0.9412,
                  "n": 82,
                  "highlight": false
                },
                {
                  "label": "Claude Sonnet 5.5 (low) · Claude Code",
                  "value": 0.939,
                  "lo": 0.8651,
                  "hi": 0.9737,
                  "n": 82,
                  "highlight": false
                }
              ]
            },
            {
              "name": "Unseen holdout",
              "points": [
                {
                  "label": "Jev 1.13 (TypeSafe)",
                  "value": 0.8214,
                  "lo": 0.7016,
                  "hi": 0.9,
                  "n": 56,
                  "highlight": true
                },
                {
                  "label": "Claude Haiku 4.5 · Claude Code",
                  "value": 0.7857,
                  "lo": 0.6618,
                  "hi": 0.8729,
                  "n": 56,
                  "highlight": false
                },
                {
                  "label": "Claude Sonnet 5.5 (low) · Claude Code",
                  "value": 0.875,
                  "lo": 0.7637,
                  "hi": 0.9381,
                  "n": 56,
                  "highlight": false
                }
              ]
            }
          ],
          "note": "Whiskers are 95% Wilson intervals. The two case sets differ in mix and size, so a gap mixes a change of case set with any change in the router, and the data cannot separate them. A gap counts only when the two intervals do not overlap.",
          "whisker": "ci95",
          "sourceIds": [
            "agent-routing-holdout",
            "agent-routing"
          ]
        },
        {
          "id": "routing-holdout-latency",
          "title": "Time per routing decision, by route",
          "subtitle": "Median, whisker to the 95th percentile",
          "kind": "dot-range",
          "unit": "seconds",
          "yLabel": "Seconds",
          "series": [
            {
              "name": "Wall time",
              "points": [
                {
                  "label": "Jev 1.13 (TypeSafe)",
                  "value": 0.139,
                  "lo": 0.139,
                  "hi": 0.192,
                  "n": 168,
                  "highlight": true
                },
                {
                  "label": "Claude Haiku 4.5 · Claude Code",
                  "value": 9.444,
                  "lo": 9.444,
                  "hi": 25.413,
                  "n": 56,
                  "highlight": false
                },
                {
                  "label": "Claude Sonnet 5.5 (low) · Claude Code",
                  "value": 2.359,
                  "lo": 2.359,
                  "hi": 3.657,
                  "n": 56,
                  "highlight": false
                }
              ]
            },
            {
              "name": "Model time (API, CLI-reported)",
              "points": [
                {
                  "label": "Claude Haiku 4.5 · Claude Code",
                  "value": 7.522,
                  "lo": 7.522,
                  "hi": 23.913,
                  "n": 56
                },
                {
                  "label": "Claude Sonnet 5.5 (low) · Claude Code",
                  "value": 1.485,
                  "lo": 1.485,
                  "hi": 2.377,
                  "n": 56
                }
              ]
            }
          ],
          "note": "The routes differ. Jev: one HTTPS call to the vendor API from one Mac, all three repetitions. Claude: the Claude Code CLI from process start to exit, which adds start-up time a direct API call would not. The whisker runs from the median to the 95th percentile; it is not a confidence interval. The shared Mac also ran local arena model servers from 15:18 local time (20:18 UTC), before all holdout calls. Host contention may affect wall times, especially CLI times. We did not rerun the timing.",
          "whisker": "p50-p95",
          "sourceIds": [
            "agent-routing-holdout"
          ]
        },
        {
          "id": "routing-holdout-cost-per-1000",
          "title": "Cost per 1,000 unseen routing decisions",
          "subtitle": "Reported tokens per decision × list price",
          "kind": "bar",
          "unit": "usd",
          "yLabel": "USD per 1,000 decisions",
          "series": [
            {
              "name": "Cost",
              "points": [
                {
                  "label": "Jev 1.13 (TypeSafe)",
                  "value": 0.03065,
                  "n": 168,
                  "highlight": true
                },
                {
                  "label": "Claude Haiku 4.5 · Claude Code",
                  "value": 7.129,
                  "n": 56,
                  "highlight": false
                },
                {
                  "label": "Claude Sonnet 5.5 (low) · Claude Code",
                  "value": 7.244,
                  "n": 56,
                  "highlight": false
                }
              ]
            }
          ],
          "note": "Calculation, not a bill. Jev: reported input tokens × $0.042 per million, output free. Claude: CLI-reported tokens × list price; the CLI wrote its prompt cache as 1-hour writes, priced at 2× input, and adds its own system prompt and tool-schema tokens. The Claude routers ran on a subscription. At the tuned-set study’s convention (every cache write at 1.25× input), Sonnet 5.5 (low) would be $4.877 here. That study shows $4.996 for it on its own cases, where the CLI reported $7.324, so its cost row and this one differ by convention and by case mix.",
          "sourceIds": [
            "agent-routing-holdout",
            "calc-repricing",
            "price-jev",
            "price-anthropic"
          ]
        }
      ],
      "tables": [
        {
          "id": "routing-holdout-pairs",
          "title": "Same unseen cases, two routers: exact McNemar test",
          "columns": [
            {
              "key": "pair",
              "label": "Pair",
              "unit": "text"
            },
            {
              "key": "cases",
              "label": "Cases",
              "unit": "count"
            },
            {
              "key": "bothRight",
              "label": "Both right",
              "unit": "count"
            },
            {
              "key": "onlyA",
              "label": "Only first right",
              "unit": "count"
            },
            {
              "key": "onlyB",
              "label": "Only second right",
              "unit": "count"
            },
            {
              "key": "bothWrong",
              "label": "Both wrong",
              "unit": "count"
            },
            {
              "key": "p",
              "label": "Exact McNemar p"
            }
          ],
          "rows": [
            {
              "pair": "Jev 1.13 (TypeSafe) vs Claude Haiku 4.5 · Claude Code",
              "cases": 56,
              "bothRight": 42,
              "onlyA": 4,
              "onlyB": 2,
              "bothWrong": 8,
              "p": 0.688
            },
            {
              "pair": "Jev 1.13 (TypeSafe) vs Claude Sonnet 5.5 (low) · Claude Code",
              "cases": 56,
              "bothRight": 42,
              "onlyA": 4,
              "onlyB": 7,
              "bothWrong": 3,
              "p": 0.549
            },
            {
              "pair": "Claude Haiku 4.5 · Claude Code vs Claude Sonnet 5.5 (low) · Claude Code",
              "cases": 56,
              "bothRight": 42,
              "onlyA": 2,
              "onlyB": 7,
              "bothWrong": 5,
              "p": 0.18
            }
          ]
        },
        {
          "id": "routing-holdout-label-agreement",
          "title": "Label agreement: the author vs a second, blind labeller (GPT-6.1 Sol)",
          "columns": [
            {
              "key": "decision",
              "label": "Decision",
              "unit": "text"
            },
            {
              "key": "question",
              "label": "Question",
              "unit": "text"
            },
            {
              "key": "n",
              "label": "Labelled",
              "unit": "count"
            },
            {
              "key": "agree",
              "label": "Second label inside our acceptable set (95% Wilson interval)",
              "unit": "text"
            },
            {
              "key": "first",
              "label": "Second label equals our first label (95% Wilson interval)",
              "unit": "text"
            },
            {
              "key": "kappa",
              "label": "Set-aware kappa (calculation; adaptive author label)",
              "unit": "score"
            },
            {
              "key": "kappaStrict",
              "label": "Cohen’s kappa, fixed first label (calculation)",
              "unit": "score"
            }
          ],
          "rows": [
            {
              "decision": "Failure class",
              "question": "failure",
              "n": 14,
              "agree": "14/14 (78.5% to 100.0%)",
              "first": "14/14 (78.5% to 100.0%)",
              "kappa": 1,
              "kappaStrict": 1
            },
            {
              "decision": "Message intent",
              "question": "intent",
              "n": 14,
              "agree": "14/14 (78.5% to 100.0%)",
              "first": "14/14 (78.5% to 100.0%)",
              "kappa": 1,
              "kappaStrict": 1
            },
            {
              "decision": "Is it a rule?",
              "question": "kind",
              "n": 14,
              "agree": "14/14 (78.5% to 100.0%)",
              "first": "14/14 (78.5% to 100.0%)",
              "kappa": 1,
              "kappaStrict": 1
            },
            {
              "decision": "Context shape",
              "question": "artifacts",
              "n": 9,
              "agree": "1/9 (2.0% to 43.5%)",
              "first": "1/9 (2.0% to 43.5%)",
              "kappa": 0,
              "kappaStrict": 0
            },
            {
              "decision": "Context shape",
              "question": "knowledge",
              "n": 14,
              "agree": "14/14 (78.5% to 100.0%)",
              "first": "11/14 (52.4% to 92.4%)",
              "kappa": 1,
              "kappaStrict": 0.567
            },
            {
              "decision": "Context shape",
              "question": "memories",
              "n": 10,
              "agree": "10/10 (72.2% to 100.0%)",
              "first": "10/10 (72.2% to 100.0%)",
              "kappa": 1,
              "kappaStrict": 1
            },
            {
              "decision": "Context shape",
              "question": "examples",
              "n": 13,
              "agree": "11/13 (57.8% to 95.7%)",
              "first": "11/13 (57.8% to 95.7%)",
              "kappa": 0.683,
              "kappaStrict": 0.683
            },
            {
              "decision": "Context shape",
              "question": "scope",
              "n": 14,
              "agree": "14/14 (78.5% to 100.0%)",
              "first": "10/14 (45.4% to 88.3%)",
              "kappa": 1,
              "kappaStrict": 0.533
            },
            {
              "decision": "Context shape",
              "question": "complexity",
              "n": 14,
              "agree": "14/14 (78.5% to 100.0%)",
              "first": "10/14 (45.4% to 88.3%)",
              "kappa": 1,
              "kappaStrict": 0.462
            },
            {
              "decision": "Context shape",
              "question": "turn",
              "n": 4,
              "agree": "4/4 (51.0% to 100.0%)",
              "first": "3/4 (30.1% to 95.4%)",
              "kappa": 1,
              "kappaStrict": 0.556
            },
            {
              "decision": "Context shape",
              "question": "transcript",
              "n": 5,
              "agree": "5/5 (56.6% to 100.0%)",
              "first": "2/5 (11.8% to 76.9%)",
              "kappa": 1,
              "kappaStrict": 0.25
            }
          ]
        },
        {
          "id": "routing-holdout-tuned-by-type",
          "title": "Tuned set vs unseen holdout, by decision group: exact rate with 95% intervals",
          "columns": [
            {
              "key": "router",
              "label": "Router",
              "unit": "text"
            },
            {
              "key": "group",
              "label": "Decision group",
              "unit": "text"
            },
            {
              "key": "tuned",
              "label": "Tuned set",
              "unit": "text"
            },
            {
              "key": "holdout",
              "label": "Unseen holdout",
              "unit": "text"
            },
            {
              "key": "change",
              "label": "Change (points, calculation)",
              "unit": "text"
            },
            {
              "key": "intervals",
              "label": "Intervals",
              "unit": "text"
            }
          ],
          "rows": [
            {
              "router": "Jev 1.13 (TypeSafe)",
              "group": "The 3 one-question types (failure class, message intent and is it a rule?)",
              "tuned": "50/50, 100.0% (92.9% to 100.0%)",
              "holdout": "38/42, 90.5% (77.9% to 96.2%)",
              "change": "−9.5",
              "intervals": "overlap"
            },
            {
              "router": "Jev 1.13 (TypeSafe)",
              "group": "Context shape (several questions per case)",
              "tuned": "24/32, 75.0% (57.9% to 86.7%)",
              "holdout": "8/14, 57.1% (32.6% to 78.6%)",
              "change": "−17.9",
              "intervals": "overlap"
            },
            {
              "router": "Claude Haiku 4.5 · Claude Code",
              "group": "The 3 one-question types (failure class, message intent and is it a rule?)",
              "tuned": "49/50, 98.0% (89.5% to 99.6%)",
              "holdout": "39/42, 92.9% (81.0% to 97.5%)",
              "change": "−5.1",
              "intervals": "overlap"
            },
            {
              "router": "Claude Haiku 4.5 · Claude Code",
              "group": "Context shape (several questions per case)",
              "tuned": "24/32, 75.0% (57.9% to 86.7%)",
              "holdout": "5/14, 35.7% (16.3% to 61.2%)",
              "change": "−39.3",
              "intervals": "overlap"
            },
            {
              "router": "Claude Sonnet 5.5 (low) · Claude Code",
              "group": "The 3 one-question types (failure class, message intent and is it a rule?)",
              "tuned": "50/50, 100.0% (92.9% to 100.0%)",
              "holdout": "41/42, 97.6% (87.7% to 99.6%)",
              "change": "−2.4",
              "intervals": "overlap"
            },
            {
              "router": "Claude Sonnet 5.5 (low) · Claude Code",
              "group": "Context shape (several questions per case)",
              "tuned": "27/32, 84.4% (68.2% to 93.1%)",
              "holdout": "8/14, 57.1% (32.6% to 78.6%)",
              "change": "−27.2",
              "intervals": "overlap"
            }
          ]
        },
        {
          "id": "routing-holdout-cases",
          "title": "Every holdout case: labels and who answered it exactly",
          "columns": [
            {
              "key": "case",
              "label": "Case",
              "unit": "text"
            },
            {
              "key": "decision",
              "label": "Decision",
              "unit": "text"
            },
            {
              "key": "labels",
              "label": "Acceptable answers",
              "unit": "text"
            },
            {
              "key": "agreed",
              "label": "Keys both labellers accept",
              "unit": "text"
            },
            {
              "key": "jev",
              "label": "Jev 1.13 (TypeSafe)",
              "unit": "text"
            },
            {
              "key": "claude-haiku",
              "label": "Claude Haiku 4.5 · Claude Code",
              "unit": "text"
            },
            {
              "key": "claude-sonnet",
              "label": "Claude Sonnet 5.5 (low) · Claude Code",
              "unit": "text"
            },
            {
              "key": "jevReps",
              "label": "Jev exact in 3 repetitions",
              "unit": "text"
            }
          ],
          "rows": [
            {
              "case": "holdout-01",
              "decision": "Failure class",
              "labels": "failure: transient",
              "agreed": "1/1",
              "jev": "exact",
              "claude-haiku": "exact",
              "claude-sonnet": "exact",
              "jevReps": "3/3"
            },
            {
              "case": "holdout-02",
              "decision": "Failure class",
              "labels": "failure: transient",
              "agreed": "1/1",
              "jev": "exact",
              "claude-haiku": "exact",
              "claude-sonnet": "exact",
              "jevReps": "3/3"
            },
            {
              "case": "holdout-03",
              "decision": "Failure class",
              "labels": "failure: transient",
              "agreed": "1/1",
              "jev": "exact",
              "claude-haiku": "exact",
              "claude-sonnet": "exact",
              "jevReps": "3/3"
            },
            {
              "case": "holdout-04",
              "decision": "Failure class",
              "labels": "failure: impossible-here",
              "agreed": "1/1",
              "jev": "exact",
              "claude-haiku": "exact",
              "claude-sonnet": "exact",
              "jevReps": "3/3"
            },
            {
              "case": "holdout-05",
              "decision": "Failure class",
              "labels": "failure: impossible-here",
              "agreed": "1/1",
              "jev": "exact",
              "claude-haiku": "exact",
              "claude-sonnet": "exact",
              "jevReps": "3/3"
            },
            {
              "case": "holdout-06",
              "decision": "Failure class",
              "labels": "failure: impossible-here",
              "agreed": "1/1",
              "jev": "exact",
              "claude-haiku": "exact",
              "claude-sonnet": "exact",
              "jevReps": "3/3"
            },
            {
              "case": "holdout-07",
              "decision": "Failure class",
              "labels": "failure: impossible-here",
              "agreed": "1/1",
              "jev": "exact",
              "claude-haiku": "exact",
              "claude-sonnet": "exact",
              "jevReps": "3/3"
            },
            {
              "case": "holdout-08",
              "decision": "Failure class",
              "labels": "failure: flaky",
              "agreed": "1/1",
              "jev": "exact",
              "claude-haiku": "exact",
              "claude-sonnet": "exact",
              "jevReps": "3/3"
            },
            {
              "case": "holdout-09",
              "decision": "Failure class",
              "labels": "failure: flaky",
              "agreed": "1/1",
              "jev": "0/1 keys",
              "claude-haiku": "0/1 keys",
              "claude-sonnet": "exact",
              "jevReps": "1/3"
            },
            {
              "case": "holdout-10",
              "decision": "Failure class",
              "labels": "failure: flaky",
              "agreed": "1/1",
              "jev": "exact",
              "claude-haiku": "exact",
              "claude-sonnet": "exact",
              "jevReps": "3/3"
            },
            {
              "case": "holdout-11",
              "decision": "Failure class",
              "labels": "failure: real",
              "agreed": "1/1",
              "jev": "exact",
              "claude-haiku": "exact",
              "claude-sonnet": "exact",
              "jevReps": "3/3"
            },
            {
              "case": "holdout-12",
              "decision": "Failure class",
              "labels": "failure: real",
              "agreed": "1/1",
              "jev": "exact",
              "claude-haiku": "exact",
              "claude-sonnet": "exact",
              "jevReps": "3/3"
            },
            {
              "case": "holdout-13",
              "decision": "Failure class",
              "labels": "failure: real",
              "agreed": "1/1",
              "jev": "exact",
              "claude-haiku": "exact",
              "claude-sonnet": "exact",
              "jevReps": "3/3"
            },
            {
              "case": "holdout-14",
              "decision": "Failure class",
              "labels": "failure: real",
              "agreed": "1/1",
              "jev": "exact",
              "claude-haiku": "exact",
              "claude-sonnet": "exact",
              "jevReps": "3/3"
            },
            {
              "case": "holdout-15",
              "decision": "Message intent",
              "labels": "intent: request",
              "agreed": "1/1",
              "jev": "exact",
              "claude-haiku": "exact",
              "claude-sonnet": "exact",
              "jevReps": "3/3"
            },
            {
              "case": "holdout-16",
              "decision": "Message intent",
              "labels": "intent: request",
              "agreed": "1/1",
              "jev": "exact",
              "claude-haiku": "exact",
              "claude-sonnet": "exact",
              "jevReps": "3/3"
            },
            {
              "case": "holdout-17",
              "decision": "Message intent",
              "labels": "intent: request",
              "agreed": "1/1",
              "jev": "exact",
              "claude-haiku": "exact",
              "claude-sonnet": "exact",
              "jevReps": "3/3"
            },
            {
              "case": "holdout-18",
              "decision": "Message intent",
              "labels": "intent: followup",
              "agreed": "1/1",
              "jev": "exact",
              "claude-haiku": "exact",
              "claude-sonnet": "exact",
              "jevReps": "3/3"
            },
            {
              "case": "holdout-19",
              "decision": "Message intent",
              "labels": "intent: followup",
              "agreed": "1/1",
              "jev": "exact",
              "claude-haiku": "exact",
              "claude-sonnet": "exact",
              "jevReps": "3/3"
            },
            {
              "case": "holdout-20",
              "decision": "Message intent",
              "labels": "intent: followup",
              "agreed": "1/1",
              "jev": "exact",
              "claude-haiku": "exact",
              "claude-sonnet": "exact",
              "jevReps": "3/3"
            },
            {
              "case": "holdout-21",
              "decision": "Message intent",
              "labels": "intent: status",
              "agreed": "1/1",
              "jev": "exact",
              "claude-haiku": "exact",
              "claude-sonnet": "exact",
              "jevReps": "3/3"
            },
            {
              "case": "holdout-22",
              "decision": "Message intent",
              "labels": "intent: status",
              "agreed": "1/1",
              "jev": "exact",
              "claude-haiku": "exact",
              "claude-sonnet": "exact",
              "jevReps": "3/3"
            },
            {
              "case": "holdout-23",
              "decision": "Message intent",
              "labels": "intent: status",
              "agreed": "1/1",
              "jev": "exact",
              "claude-haiku": "exact",
              "claude-sonnet": "exact",
              "jevReps": "3/3"
            },
            {
              "case": "holdout-24",
              "decision": "Message intent",
              "labels": "intent: decision",
              "agreed": "1/1",
              "jev": "exact",
              "claude-haiku": "exact",
              "claude-sonnet": "exact",
              "jevReps": "3/3"
            },
            {
              "case": "holdout-25",
              "decision": "Message intent",
              "labels": "intent: decision",
              "agreed": "1/1",
              "jev": "0/1 keys",
              "claude-haiku": "exact",
              "claude-sonnet": "exact",
              "jevReps": "0/3"
            },
            {
              "case": "holdout-26",
              "decision": "Message intent",
              "labels": "intent: control",
              "agreed": "1/1",
              "jev": "exact",
              "claude-haiku": "exact",
              "claude-sonnet": "exact",
              "jevReps": "3/3"
            },
            {
              "case": "holdout-27",
              "decision": "Message intent",
              "labels": "intent: control",
              "agreed": "1/1",
              "jev": "exact",
              "claude-haiku": "exact",
              "claude-sonnet": "exact",
              "jevReps": "3/3"
            },
            {
              "case": "holdout-28",
              "decision": "Message intent",
              "labels": "intent: control",
              "agreed": "1/1",
              "jev": "0/1 keys",
              "claude-haiku": "0/1 keys",
              "claude-sonnet": "exact",
              "jevReps": "0/3"
            },
            {
              "case": "holdout-29",
              "decision": "Is it a rule?",
              "labels": "kind: obligation",
              "agreed": "1/1",
              "jev": "exact",
              "claude-haiku": "exact",
              "claude-sonnet": "exact",
              "jevReps": "3/3"
            },
            {
              "case": "holdout-30",
              "decision": "Is it a rule?",
              "labels": "kind: obligation",
              "agreed": "1/1",
              "jev": "exact",
              "claude-haiku": "exact",
              "claude-sonnet": "exact",
              "jevReps": "3/3"
            },
            {
              "case": "holdout-31",
              "decision": "Is it a rule?",
              "labels": "kind: obligation",
              "agreed": "1/1",
              "jev": "exact",
              "claude-haiku": "exact",
              "claude-sonnet": "exact",
              "jevReps": "3/3"
            },
            {
              "case": "holdout-32",
              "decision": "Is it a rule?",
              "labels": "kind: obligation",
              "agreed": "1/1",
              "jev": "0/1 keys",
              "claude-haiku": "0/1 keys",
              "claude-sonnet": "0/1 keys",
              "jevReps": "0/3"
            },
            {
              "case": "holdout-33",
              "decision": "Is it a rule?",
              "labels": "kind: prohibition",
              "agreed": "1/1",
              "jev": "exact",
              "claude-haiku": "exact",
              "claude-sonnet": "exact",
              "jevReps": "3/3"
            },
            {
              "case": "holdout-34",
              "decision": "Is it a rule?",
              "labels": "kind: prohibition or obligation",
              "agreed": "1/1",
              "jev": "exact",
              "claude-haiku": "exact",
              "claude-sonnet": "exact",
              "jevReps": "3/3"
            },
            {
              "case": "holdout-35",
              "decision": "Is it a rule?",
              "labels": "kind: prohibition",
              "agreed": "1/1",
              "jev": "exact",
              "claude-haiku": "exact",
              "claude-sonnet": "exact",
              "jevReps": "3/3"
            },
            {
              "case": "holdout-36",
              "decision": "Is it a rule?",
              "labels": "kind: preference",
              "agreed": "1/1",
              "jev": "exact",
              "claude-haiku": "exact",
              "claude-sonnet": "exact",
              "jevReps": "3/3"
            },
            {
              "case": "holdout-37",
              "decision": "Is it a rule?",
              "labels": "kind: preference",
              "agreed": "1/1",
              "jev": "exact",
              "claude-haiku": "exact",
              "claude-sonnet": "exact",
              "jevReps": "3/3"
            },
            {
              "case": "holdout-38",
              "decision": "Is it a rule?",
              "labels": "kind: preference",
              "agreed": "1/1",
              "jev": "exact",
              "claude-haiku": "exact",
              "claude-sonnet": "exact",
              "jevReps": "3/3"
            },
            {
              "case": "holdout-39",
              "decision": "Is it a rule?",
              "labels": "kind: not-a-rule",
              "agreed": "1/1",
              "jev": "exact",
              "claude-haiku": "exact",
              "claude-sonnet": "exact",
              "jevReps": "3/3"
            },
            {
              "case": "holdout-40",
              "decision": "Is it a rule?",
              "labels": "kind: not-a-rule",
              "agreed": "1/1",
              "jev": "exact",
              "claude-haiku": "exact",
              "claude-sonnet": "exact",
              "jevReps": "3/3"
            },
            {
              "case": "holdout-41",
              "decision": "Is it a rule?",
              "labels": "kind: not-a-rule",
              "agreed": "1/1",
              "jev": "exact",
              "claude-haiku": "exact",
              "claude-sonnet": "exact",
              "jevReps": "3/3"
            },
            {
              "case": "holdout-42",
              "decision": "Is it a rule?",
              "labels": "kind: not-a-rule",
              "agreed": "1/1",
              "jev": "exact",
              "claude-haiku": "exact",
              "claude-sonnet": "exact",
              "jevReps": "3/3"
            },
            {
              "case": "holdout-43",
              "decision": "Context shape",
              "labels": "artifacts: no; knowledge: search; memories: yes; examples: yes; scope: small or medium; complexity: simple or moderate",
              "agreed": "5/6",
              "jev": "exact",
              "claude-haiku": "5/6 keys",
              "claude-sonnet": "5/6 keys",
              "jevReps": "3/3"
            },
            {
              "case": "holdout-44",
              "decision": "Context shape",
              "labels": "artifacts: no; knowledge: search; memories: yes; examples: no; scope: medium; complexity: moderate or complex",
              "agreed": "5/6",
              "jev": "exact",
              "claude-haiku": "4/6 keys",
              "claude-sonnet": "exact",
              "jevReps": "3/3"
            },
            {
              "case": "holdout-45",
              "decision": "Context shape",
              "labels": "turn: follow-up; transcript: recent or full; knowledge: search; scope: small or medium; complexity: moderate",
              "agreed": "5/5",
              "jev": "exact",
              "claude-haiku": "exact",
              "claude-sonnet": "exact",
              "jevReps": "3/3"
            },
            {
              "case": "holdout-46",
              "decision": "Context shape",
              "labels": "turn: clarification; transcript: recent or full; knowledge: search; examples: no; scope: small or medium; complexity: moderate",
              "agreed": "6/6",
              "jev": "5/6 keys",
              "claude-haiku": "5/6 keys",
              "claude-sonnet": "exact",
              "jevReps": "0/3"
            },
            {
              "case": "holdout-47",
              "decision": "Context shape",
              "labels": "transcript: summary or recent; knowledge: search; memories: yes; examples: no; scope: medium or large; complexity: moderate or complex",
              "agreed": "5/6",
              "jev": "exact",
              "claude-haiku": "exact",
              "claude-sonnet": "5/6 keys",
              "jevReps": "3/3"
            },
            {
              "case": "holdout-48",
              "decision": "Context shape",
              "labels": "artifacts: no; knowledge: none; memories: no; examples: no; scope: small; complexity: simple or moderate",
              "agreed": "5/6",
              "jev": "5/6 keys",
              "claude-haiku": "4/6 keys",
              "claude-sonnet": "exact",
              "jevReps": "0/3"
            },
            {
              "case": "holdout-49",
              "decision": "Context shape",
              "labels": "artifacts: no; knowledge: search; memories: yes; examples: yes; scope: small or medium; complexity: moderate",
              "agreed": "5/6",
              "jev": "5/6 keys",
              "claude-haiku": "exact",
              "claude-sonnet": "exact",
              "jevReps": "0/3"
            },
            {
              "case": "holdout-50",
              "decision": "Context shape",
              "labels": "artifacts: no; knowledge: search; memories: yes; examples: no; scope: medium; complexity: complex",
              "agreed": "5/6",
              "jev": "exact",
              "claude-haiku": "exact",
              "claude-sonnet": "5/6 keys",
              "jevReps": "3/3"
            },
            {
              "case": "holdout-51",
              "decision": "Context shape",
              "labels": "transcript: recent or full; knowledge: facts or search; memories: yes; examples: no; scope: medium; complexity: moderate",
              "agreed": "6/6",
              "jev": "3/6 keys",
              "claude-haiku": "2/6 keys",
              "claude-sonnet": "exact",
              "jevReps": "0/3"
            },
            {
              "case": "holdout-52",
              "decision": "Context shape",
              "labels": "artifacts: no; knowledge: facts or search; memories: yes; examples: yes; scope: medium or large; complexity: moderate",
              "agreed": "6/6",
              "jev": "5/6 keys",
              "claude-haiku": "2/6 keys",
              "claude-sonnet": "4/6 keys",
              "jevReps": "0/3"
            },
            {
              "case": "holdout-53",
              "decision": "Context shape",
              "labels": "turn: new-task or follow-up; artifacts: no; knowledge: facts or search; memories: yes; examples: no; scope: small or medium; complexity: moderate",
              "agreed": "6/7",
              "jev": "exact",
              "claude-haiku": "5/7 keys",
              "claude-sonnet": "exact",
              "jevReps": "3/3"
            },
            {
              "case": "holdout-54",
              "decision": "Context shape",
              "labels": "turn: follow-up; transcript: recent or full; knowledge: facts or search; examples: yes; scope: small or medium; complexity: moderate",
              "agreed": "6/6",
              "jev": "exact",
              "claude-haiku": "exact",
              "claude-sonnet": "exact",
              "jevReps": "3/3"
            },
            {
              "case": "holdout-55",
              "decision": "Context shape",
              "labels": "artifacts: no; knowledge: search; memories: yes; examples: no; scope: large; complexity: complex",
              "agreed": "4/6",
              "jev": "exact",
              "claude-haiku": "5/6 keys",
              "claude-sonnet": "5/6 keys",
              "jevReps": "3/3"
            },
            {
              "case": "holdout-56",
              "decision": "Context shape",
              "labels": "artifacts: no; knowledge: facts; examples: no; scope: small; complexity: simple or moderate",
              "agreed": "4/5",
              "jev": "4/5 keys",
              "claude-haiku": "2/5 keys",
              "claude-sonnet": "2/5 keys",
              "jevReps": "0/3"
            }
          ]
        },
        {
          "id": "routing-holdout-routers",
          "title": "Every router on the holdout",
          "columns": [
            {
              "key": "router",
              "label": "Router",
              "unit": "text"
            },
            {
              "key": "route",
              "label": "Route",
              "unit": "text"
            },
            {
              "key": "calls",
              "label": "Counted calls",
              "unit": "count"
            },
            {
              "key": "exact",
              "label": "Exact",
              "unit": "text"
            },
            {
              "key": "keys",
              "label": "Key accuracy",
              "unit": "text"
            },
            {
              "key": "agreed",
              "label": "Exact, agreed keys only",
              "unit": "text"
            },
            {
              "key": "tokens",
              "label": "Mean tokens per decision (calculation; input incl. cache / output)",
              "unit": "text"
            },
            {
              "key": "cost",
              "label": "USD per 1,000 (calculation)",
              "unit": "usd"
            }
          ],
          "rows": [
            {
              "router": "Jev 1.13 (TypeSafe)",
              "route": "direct HTTPS to the vendor API with the production request body",
              "calls": 168,
              "exact": "46 of 56 (82%, 95% interval 70% to 90%)",
              "keys": "113 of 125 (90%, 95% interval 84% to 94%)",
              "agreed": "46 of 56 (82%, 95% interval 70% to 90%)",
              "tokens": "730 / 119.3",
              "cost": 0.03065
            },
            {
              "router": "Claude Haiku 4.5 · Claude Code",
              "route": "Claude Code CLI, one call per decision, CLI default effort (extended thinking)",
              "calls": 56,
              "exact": "44 of 56 (79%, 95% interval 66% to 87%)",
              "keys": "102 of 125 (82%, 95% interval 74% to 87%)",
              "agreed": "46 of 56 (82%, 95% interval 70% to 90%)",
              "tokens": "1,775 / 1,071",
              "cost": 7.129
            },
            {
              "router": "Claude Sonnet 5.5 (low) · Claude Code",
              "route": "Claude Code CLI, one call per decision, effort low",
              "calls": 56,
              "exact": "49 of 56 (88%, 95% interval 76% to 94%)",
              "keys": "115 of 125 (92%, 95% interval 86% to 96%)",
              "agreed": "52 of 56 (93%, 95% interval 83% to 97%)",
              "tokens": "1,705 / 90.5",
              "cost": 7.244
            }
          ]
        }
      ],
      "related": [
        "routing-jev-vs-llm",
        "routing-overhead"
      ]
    },
    {
      "slug": "thinking-token-bill",
      "title": "How much of an AI bill is thinking? Reasoning tokens by model and effort",
      "seoTitle": "Thinking tokens by model and effort: share and cost",
      "description": "Reasoning tokens in 378 recorded calls by model and effort: share of output, list-price cost per call and per pass, and time. Calculations.",
      "question": "Across 378 recorded calls of Haiku 4.5, Sonnet 5.5, Opus 5.5, Fable 5.1 and GPT-6.1 Sol, how many output tokens come from reasoning (thinking)? What do they cost at list price per call and per strict pass, and do they track time? The calls cover eight hard tasks, five short tasks and several efforts.",
      "answer": "On eight hard tasks, reasoning tokens were 46% to 92% of the output tokens of a median call (16 to 24 calls per configuration). Haiku 4.5 had the highest median (92%), GPT-6.1 Sol (medium) had the lowest (46%) and the other 5 sat between 54% and 64%. Per-call ranges are wide (Sonnet 5.5 0% to 96%). The ranges of every pair overlap, so this run ranks no configuration. Output share is not bill share. At list price (a calculation) the reasoning part of a call cost a mean $0.0023 (GPT-6.1 Sol (medium)) to $0.0537 (Fable 5.1). That was 9% (GPT-6.1 Sol (medium)) to 80% (Haiku 4.5) of the total list-price cost in each configuration. This pooled share divides summed reasoning cost by summed total cost. Input tokens cost money too. The calls ran on flat subscriptions, so this is not a bill. Fable 5.1's thinking cost 8.1x Sonnet 5.5's per call. The output price explains a factor of 5.0 ($50 against $10 per million tokens). More reasoning tokens explain the rest, a factor of 1.6 (means). Higher effort settings had higher mean recorded reasoning counts in these batches. We paired the same eight tasks. At high effort, mean reasoning tokens per call were higher than at low effort on these tasks: Sonnet 5.5 7 of 8, Opus 5.5 8 of 8, GPT-6.1 Sol 8 of 8. The pooled reasoning share rose from low to high effort. Sonnet 5.5 53% to 73%. Opus 5.5 38% to 69%. GPT-6.1 Sol 29% to 59%. It rose at each step (low, medium, high) in all 3 ladders. The reasoning cost per call rose 2.2x for Sonnet 5.5, 3.6x for Opus 5.5 and 3.4x for GPT-6.1 Sol (a calculation). Every effort cell passed 16/16 strictly. The pass count stayed the same on this set, which has a ceiling (95% Wilson 81% to 100% per cell). Recorded reasoning counts correlated with total time. Within each Claude model, the rank correlation between reasoning tokens and total time was 0.85 to 0.98 (Spearman, 24 to 80 calls each). On the same task, 1,000 more reasoning tokens went with 7.6 to 14.5 s more time (a calculation). For GPT-6.1 Sol in the Codex CLI, the recorded rank correlation was: 0.56 (48 calls). Thinking costs money whether or not the call passes. Haiku 4.5 passed 11/24 strictly (95% Wilson 28% to 65%). The 13 calls that did not pass held 61% of its reasoning cost (a calculation).",
      "date": "2026-10-06",
      "updated": "2026-10-06",
      "tags": [
        "thought-experiment",
        "calculation",
        "reasoning-tokens",
        "thinking-tokens",
        "effort",
        "llm-pricing",
        "claude-haiku",
        "claude-sonnet",
        "claude-opus",
        "claude-fable",
        "gpt-6-1-sol",
        "claude-code",
        "codex-cli",
        "latency"
      ],
      "method": [
        "This study is a calculation over receipts that already exist. We made no new model call. The inputs are three recorded runs. The hard head-to-head gives 152 completed calls. We leave its 30 pre-inference blocked attempts out of token calculations. They remain failures of the original route attempt, not model answers. The effort ladder gives 96 new calls. The five-task head-to-head gives 130 calls. All current counted calls completed. Recorded errors count as failures when their usage is known. We keep the separate blocked batch on record. The original Codex route was blocked. A later batch ran after a successful uncounted probe. It resumed after two completed calls and skipped them.",
        "Sum check. Before we compute any share, we test one point. Do the output tokens include the reasoning tokens? We use this accounting assumption and check its consistency with the receipts. We run three tests on each CLI route. Test 1: reasoning never exceeds output. Test 2: within one task and model, one more reasoning token adds about one output token. A slope near 1 supports the assumption. Correlation alone cannot prove the counter semantics. Test 3: we compare characters per remaining token with characters per output token on calls whose reasoning counter is zero. Claude Code: consistent with inclusion, not proof. Reasoning never exceeded output (0 of 290 calls). Within one task and model, each extra reasoning token went with 0.99 extra output tokens (leave-one-task-out range 0.98 to 1.01, 290 calls). Output minus reasoning has 1.94 characters per token. Calls with 0 reasoning have 1.98. For comparison, output including reasoning has only 0.45. Codex CLI: consistent with inclusion, not proof. Reasoning never exceeded output (0 of 88 calls). Within one task and model, each extra reasoning token went with 0.97 extra output tokens (leave-one-task-out range 0.97 to 0.99, 88 calls). Output minus reasoning has 2.99 characters per token. Calls with 0 reasoning have 2.76. For comparison, output including reasoning has only 1.36.",
        "Not reported is not zero. A route reports reasoning when at least one of its calls reports more than 0. Both routes do (Claude Code 212 of 290 calls, Codex CLI 68 of 88 calls). So a reported 0 is the recorded counter, not proof of no internal reasoning. 98 calls reported 0, and their replies look like visible text only (test 3). If any model attempt lacks usable token counts, this builder returns no study. It never prices unknown usage as zero. 0 calls had no usable count.",
        "Reasoning share of one call = reasoning tokens ÷ output tokens. A configuration gets two numbers. One is the median of its per-call shares. The other is the pooled share (all reasoning tokens ÷ all output tokens). Ranges are the lowest and highest call. They are not intervals.",
        "Hard tasks: we use all calls of each configuration in the hard head-to-head. That is 16 to 24 calls: 8 tasks × 3 repetitions for Claude Code and 8 × 2 for Codex CLI. Effort ladder: 11 cells of 8 tasks × 2 repetitions, the design of the effort-ladder study. We reuse its reference cells from the hard head-to-head (Claude repetitions 1-2 only). Short tasks: the five validated tasks (10 to 15 calls per configuration).",
        "Cost is a calculation at list price, not a bill. The calls ran on flat subscriptions. Reasoning cost = reasoning tokens × the model's output price. Remaining output cost = (output tokens − reasoning tokens) × the same price. We use the rest as a visible-answer estimate. Input cost covers the whole prompt. Cache writes use the one-hour list-price assumption; the receipts do not state the cache lifetime. Prices per million output tokens: Claude Haiku 4.5 $5, Claude Sonnet 5.5 $10, Claude Opus 5.5 $20, Claude Fable 5.1 $50, GPT-6.1 Sol $10. Sources: Anthropic list prices of 2026-09-21, OpenAI of 2026-10-03.",
        "Cost per strict pass = the list-price cost of all calls in the cell, failures included, ÷ the cell's strict passes. Hard and effort cells use the hard-set strict rule. Short-task cells use their original rule, which can strip a wrapping fence.",
        "Time link: we use all calls of the hard head-to-head and the effort ladder. For each model and route, we compute the Spearman rank correlation of reasoning tokens with total time (with a leave-one-task-out sensitivity range, not a confidence interval). We also compute the least-squares slope in seconds per 1,000 reasoning tokens. Then we compute the slope within task. We centre each task on its own mean. This controls for differences in task means, not effort or batch effects.",
        "Isolation, tasks, validators and flags follow the hard head-to-head and the five-task head-to-head. Each call ran in a fresh empty folder with tools off and one turn. The timeout was 300 s per call (180 s for the short tasks). One call ran at a time per account.",
        "Claude Code calls ran with an output cap of 16,000 tokens in all three runs. The highest output of any analysed Claude call was 9,321, so no call hit the cap. The Codex CLI has no cap setting. Its highest output was 2,569."
      ],
      "caveats": [
        "The hard-set protocol file was created after its first counted call. Both effort-ladder protocol files were created after their batches ended. The short-set file predates its first call, but its top-up amendment timing is unverified. Batch receipts preserve protocol text, but we cannot verify all rules were written before inference. Treat these as exploratory calculations, not preregistered tests.",
        "The calls used one shared Mac and network. Other work and provider load were not controlled. Latency includes those effects.",
        "These are hand-built, tuned case sets, not random workload samples. The short set and most hard-set configurations hit a pass ceiling. Repeats of the same tasks are not independent. Wilson pass intervals describe counted calls under an independence assumption, not performance on new tasks.",
        "Each CLI reports its own reasoning counter, and we never see the reasoning text. So we cannot check what each vendor counts. The sum check shows only that the counters are consistent with inclusion in output; it does not prove their semantics. Shares from Claude Code and Codex CLI do not measure like for like how much each model thinks.",
        "Every prompt asks for a short reply in a strict format (code, JSON, one line or a regex, with no explanation). So the visible answer is short and the reasoning share is high. Longer visible replies could change the share. We did not measure that workload.",
        "Small cells: 16 to 24 calls per configuration over 8 tasks. Per-call ranges are wide and overlap for every pair of configurations. So the medians describe this run and rank nothing. A sentence says one side is ahead only when its ranges do not overlap.",
        "Input includes CLI context that these receipts do not separately count. It moves with cache hits. So the part of the call cost that goes to reasoning depends on the CLI and on the cache, not only on the model. GPT-6.1 Sol at medium and at high effort ran in different batches and show very different input cost per call.",
        "Every effort-ladder cell passed 16/16, so the set has a ceiling. Higher effort had higher mean recorded reasoning cost in these batches. It does not show that more thinking never helps on harder work.",
        "The time link is a correlation, not a cause. Task, effort and batch can affect reasoning, visible output and time together. Total time includes CLI start-up. The within-task slope controls for differences in task means. It does not remove effort or batch effects. Output tokens (reasoning plus visible answer) track time at least as closely as reasoning alone. The Codex CLI slope (36.4 s per 1,000 reasoning tokens) has a calculated ratio of 2.5 to 4.8 times the Claude Code slopes. We did not test why. Calls are repeats of 8 tasks, so they are not independent. The ranges omit one whole task at a time. They show sensitivity to the task mix, not 95% coverage.",
        "Effort levels are not the same scale across vendors. \"Default\" means we did not pass the effort flag, and the CLI chose. The effort-ladder reference cells ran in a different batch and hour than the new cells.",
        "List-price costs are calculations, because the calls used flat subscriptions. A price change moves every cost here. It leaves every token count unchanged."
      ],
      "sourceIds": [
        "calc-thinking-bill",
        "agent-provider-h2h-hard",
        "agent-effort-ladder",
        "agent-provider-h2h",
        "price-anthropic",
        "price-openai"
      ],
      "hero": {
        "statIds": [
          "thinking-bill-share-highest",
          "thinking-bill-share-lowest"
        ]
      },
      "stats": [
        {
          "id": "thinking-bill-calls",
          "label": "Recorded calls analysed (no new calls)",
          "value": 378,
          "unit": "count",
          "display": "378 (152 hard head-to-head, 96 new effort-ladder, 130 five-task head-to-head)",
          "n": 378
        },
        {
          "id": "thinking-bill-zero-reasoning",
          "label": "Calls that reported 0 reasoning tokens (kept as 0, not \"not reported\")",
          "value": 98,
          "unit": "count",
          "display": "98 of 378; 0 calls had no usable count",
          "n": 378
        },
        {
          "id": "thinking-bill-sum-claude",
          "label": "Sum check, Claude Code: output tokens gained per extra reasoning token, same task and model (calculation)",
          "value": 0.993,
          "unit": "ratio",
          "display": "0.99 (leave-one-task-out range 0.98 to 1.01; 290 calls)",
          "n": 290,
          "note": "About 1 is consistent with reasoning inside output. This correlation does not prove how a CLI counts tokens."
        },
        {
          "id": "thinking-bill-sum-codex",
          "label": "Sum check, Codex CLI: output tokens gained per extra reasoning token, same task and model (calculation)",
          "value": 0.967,
          "unit": "ratio",
          "display": "0.97 (leave-one-task-out range 0.97 to 0.99; 88 calls)",
          "n": 88,
          "note": "About 1 is consistent with reasoning inside output. This correlation does not prove how a CLI counts tokens."
        },
        {
          "id": "thinking-bill-share-highest",
          "label": "Highest median reasoning share of output tokens, hard tasks (calculation)",
          "value": 0.9168,
          "unit": "rate",
          "display": "92% (Claude Haiku 4.5 · Claude Code; range 76% to 99%)",
          "n": 24
        },
        {
          "id": "thinking-bill-share-lowest",
          "label": "Lowest median reasoning share of output tokens, hard tasks (calculation)",
          "value": 0.4633,
          "unit": "rate",
          "display": "46% (GPT-6.1 Sol (medium) · Codex CLI; range 11% to 87%)",
          "n": 16
        },
        {
          "id": "thinking-bill-reasoning-cost-highest",
          "label": "Highest mean list-price cost of reasoning per call, hard tasks (calculation)",
          "value": 0.053696,
          "unit": "usd",
          "display": "$0.0537 (Claude Fable 5.1 · Claude Code; call range $0.0037 to $0.2944)",
          "n": 24
        },
        {
          "id": "thinking-bill-reasoning-cost-lowest",
          "label": "Lowest mean list-price cost of reasoning per call, hard tasks (calculation)",
          "value": 0.002273,
          "unit": "usd",
          "display": "$0.0023 (GPT-6.1 Sol (medium) · Codex CLI; call range $0.0006 to $0.0084)",
          "n": 16
        },
        {
          "id": "thinking-bill-cost-share-highest",
          "label": "Highest pooled reasoning share of total list-price cost, hard tasks (calculation)",
          "value": 0.7953,
          "unit": "rate",
          "display": "80% (Claude Haiku 4.5 · Claude Code; call range 40% to 90%)",
          "n": 24
        },
        {
          "id": "thinking-bill-cost-share-lowest",
          "label": "Lowest pooled reasoning share of total list-price cost, hard tasks (calculation)",
          "value": 0.0887,
          "unit": "rate",
          "display": "9% (GPT-6.1 Sol (medium) · Codex CLI; call range 2% to 20%)",
          "n": 16
        },
        {
          "id": "thinking-bill-fable-vs-sonnet",
          "label": "Reasoning cost per call, Fable 5.1 ÷ Sonnet 5.5, hard tasks (calculation, means)",
          "value": 8.056,
          "unit": "ratio",
          "display": "8.1x ($0.0537 vs $0.0067)",
          "n": 24
        },
        {
          "id": "thinking-bill-haiku-failed-share",
          "label": "Share of Haiku 4.5 reasoning cost spent on calls that did not pass (calculation)",
          "value": 0.6114,
          "unit": "rate",
          "display": "61% (13 of 24 calls did not pass strictly)",
          "n": 24
        },
        {
          "id": "thinking-bill-effort-tasks-sonnet-5-5",
          "label": "Tasks where mean reasoning tokens per call were higher at high than at low effort, Claude Sonnet 5.5 · Claude Code (paired, same tasks; calculation)",
          "value": 7,
          "unit": "count",
          "display": "7 of 8 tasks",
          "n": 8,
          "note": "Each task ran 2 times at each effort. Eight tasks is a small sample; this counts tasks, it is not an interval."
        },
        {
          "id": "thinking-bill-effort-tasks-opus-5-5",
          "label": "Tasks where mean reasoning tokens per call were higher at high than at low effort, Claude Opus 5.5 · Claude Code (paired, same tasks; calculation)",
          "value": 8,
          "unit": "count",
          "display": "8 of 8 tasks",
          "n": 8,
          "note": "Each task ran 2 times at each effort. Eight tasks is a small sample; this counts tasks, it is not an interval."
        },
        {
          "id": "thinking-bill-effort-tasks-gpt-6-1-sol",
          "label": "Tasks where mean reasoning tokens per call were higher at high than at low effort, GPT-6.1 Sol · Codex CLI (paired, same tasks; calculation)",
          "value": 8,
          "unit": "count",
          "display": "8 of 8 tasks",
          "n": 8,
          "note": "Each task ran 2 times at each effort. Eight tasks is a small sample; this counts tasks, it is not an interval."
        },
        {
          "id": "thinking-bill-effort-ratio-sonnet-5-5",
          "label": "Reasoning cost per call, high ÷ low effort, Claude Sonnet 5.5 · Claude Code (calculation, means)",
          "value": 2.17,
          "unit": "ratio",
          "display": "2.2x ($0.0043 at low, $0.0094 at high)",
          "n": 16
        },
        {
          "id": "thinking-bill-effort-ratio-opus-5-5",
          "label": "Reasoning cost per call, high ÷ low effort, Claude Opus 5.5 · Claude Code (calculation, means)",
          "value": 3.584,
          "unit": "ratio",
          "display": "3.6x ($0.0050 at low, $0.0180 at high)",
          "n": 16
        },
        {
          "id": "thinking-bill-effort-ratio-gpt-6-1-sol",
          "label": "Reasoning cost per call, high ÷ low effort, GPT-6.1 Sol · Codex CLI (calculation, means)",
          "value": 3.366,
          "unit": "ratio",
          "display": "3.4x ($0.0012 at low, $0.0041 at high)",
          "n": 16
        },
        {
          "id": "thinking-bill-time-rho-claude",
          "label": "Spearman, reasoning tokens vs total time, per Claude model: lowest (calculation)",
          "value": 0.852,
          "unit": "score",
          "display": "0.85 to 0.98 across 4 Claude models",
          "n": 200,
          "note": "The value is the lowest of the per-model correlations; the display gives the range."
        },
        {
          "id": "thinking-bill-time-slope-claude",
          "label": "Seconds per 1,000 reasoning tokens on the same task, per Claude model: lowest (calculation)",
          "value": 7.649,
          "unit": "seconds",
          "display": "7.6 to 14.5 s across 4 Claude models",
          "n": 200,
          "note": "The value is the lowest of the per-model slopes; the display gives the range."
        },
        {
          "id": "thinking-bill-time-rho-codex",
          "label": "Spearman, reasoning tokens vs total time, GPT-6.1 Sol in Codex CLI (calculation)",
          "value": 0.556,
          "unit": "score",
          "display": "0.56 (leave-one-task-out range 0.34 to 0.65; 48 calls)",
          "n": 48,
          "note": "The range omits one whole task at a time. It is a sensitivity check, not a confidence interval."
        }
      ],
      "charts": [
        {
          "id": "thinking-bill-share",
          "title": "Reasoning share of output tokens per call on hard tasks (calculation)",
          "subtitle": "Median call: reasoning tokens ÷ output tokens. Whiskers: lowest and highest call (16 to 24 calls per configuration)",
          "kind": "bar",
          "unit": "percent",
          "yLabel": "Reasoning share of output tokens (%)",
          "series": [
            {
              "name": "Median call",
              "points": [
                {
                  "label": "Claude Haiku 4.5 · Claude Code",
                  "value": 91.68,
                  "lo": 76.46,
                  "hi": 99.27,
                  "n": 24
                },
                {
                  "label": "Claude Fable 5.1 · Claude Code",
                  "value": 64.24,
                  "lo": 23.44,
                  "hi": 97.19,
                  "n": 24
                },
                {
                  "label": "GPT-6.1 Sol (high) · Codex CLI",
                  "value": 57.01,
                  "lo": 29.19,
                  "hi": 90.8,
                  "n": 16
                },
                {
                  "label": "Claude Opus 5.5 · Claude Code",
                  "value": 54.79,
                  "lo": 29.92,
                  "hi": 95.6,
                  "n": 24
                },
                {
                  "label": "Claude Sonnet 5.5 · Claude Code",
                  "value": 54.54,
                  "lo": 0,
                  "hi": 95.91,
                  "n": 24
                },
                {
                  "label": "Claude Opus 5.5 (high) · Claude Code",
                  "value": 54.43,
                  "lo": 36.14,
                  "hi": 96.23,
                  "n": 24
                },
                {
                  "label": "GPT-6.1 Sol (medium) · Codex CLI",
                  "value": 46.33,
                  "lo": 11.42,
                  "hi": 86.85,
                  "n": 16
                }
              ]
            }
          ],
          "note": "Calculation from reported tokens, not a run. Each call gives reasoning ÷ output; the bar is the median of those shares. Whiskers are the lowest and highest call. They are a range, not a confidence interval. They are wide, so the medians describe this run and rank nothing. The pooled share (all reasoning tokens ÷ all output tokens) is in the table. We treat reasoning tokens as part of output tokens; the consistency check supports this accounting assumption. Each CLI reports its own count.",
          "whisker": "minmax",
          "polarity": "none",
          "sourceIds": [
            "calc-thinking-bill",
            "agent-provider-h2h-hard",
            "agent-effort-ladder",
            "agent-provider-h2h",
            "price-anthropic",
            "price-openai"
          ]
        },
        {
          "id": "thinking-bill-cost-per-call",
          "title": "List-price cost per call: reasoning, remaining output and input (calculation)",
          "subtitle": "Mean per call on the hard tasks; the three parts add up to the call",
          "kind": "stacked-bar",
          "unit": "usd",
          "yLabel": "USD per call (list price)",
          "series": [
            {
              "name": "Reasoning (output tokens)",
              "points": [
                {
                  "label": "Claude Fable 5.1 · Claude Code",
                  "value": 0.053696,
                  "n": 24
                },
                {
                  "label": "Claude Opus 5.5 (high) · Claude Code",
                  "value": 0.017969,
                  "n": 24
                },
                {
                  "label": "Claude Haiku 4.5 · Claude Code",
                  "value": 0.024492,
                  "n": 24
                },
                {
                  "label": "Claude Opus 5.5 · Claude Code",
                  "value": 0.012528,
                  "n": 24
                },
                {
                  "label": "GPT-6.1 Sol (medium) · Codex CLI",
                  "value": 0.002273,
                  "n": 16
                },
                {
                  "label": "GPT-6.1 Sol (high) · Codex CLI",
                  "value": 0.004114,
                  "n": 16
                },
                {
                  "label": "Claude Sonnet 5.5 · Claude Code",
                  "value": 0.006665,
                  "n": 24
                }
              ]
            },
            {
              "name": "Remaining output (visible-answer estimate)",
              "points": [
                {
                  "label": "Claude Fable 5.1 · Claude Code",
                  "value": 0.018702,
                  "n": 24
                },
                {
                  "label": "Claude Opus 5.5 (high) · Claude Code",
                  "value": 0.00799,
                  "n": 24
                },
                {
                  "label": "Claude Haiku 4.5 · Claude Code",
                  "value": 0.001795,
                  "n": 24
                },
                {
                  "label": "Claude Opus 5.5 · Claude Code",
                  "value": 0.008003,
                  "n": 24
                },
                {
                  "label": "GPT-6.1 Sol (medium) · Codex CLI",
                  "value": 0.003025,
                  "n": 16
                },
                {
                  "label": "GPT-6.1 Sol (high) · Codex CLI",
                  "value": 0.002905,
                  "n": 16
                },
                {
                  "label": "Claude Sonnet 5.5 · Claude Code",
                  "value": 0.003672,
                  "n": 24
                }
              ]
            },
            {
              "name": "Input (prompt, cache priced)",
              "points": [
                {
                  "label": "Claude Fable 5.1 · Claude Code",
                  "value": 0.02091,
                  "n": 24
                },
                {
                  "label": "Claude Opus 5.5 (high) · Claude Code",
                  "value": 0.007407,
                  "n": 24
                },
                {
                  "label": "Claude Haiku 4.5 · Claude Code",
                  "value": 0.00451,
                  "n": 24
                },
                {
                  "label": "Claude Opus 5.5 · Claude Code",
                  "value": 0.00771,
                  "n": 24
                },
                {
                  "label": "GPT-6.1 Sol (medium) · Codex CLI",
                  "value": 0.020339,
                  "n": 16
                },
                {
                  "label": "GPT-6.1 Sol (high) · Codex CLI",
                  "value": 0.008117,
                  "n": 16
                },
                {
                  "label": "Claude Sonnet 5.5 · Claude Code",
                  "value": 0.004012,
                  "n": 24
                }
              ]
            }
          ],
          "note": "Calculation, not a bill: reported tokens × list price (per million output tokens: Haiku 4.5 $5, Sonnet 5.5 $10, Opus 5.5 $20, Fable 5.1 $50 and GPT-6.1 Sol $10); the calls ran on flat subscriptions. Reasoning and remaining output both use the output price. The remainder estimates visible-answer tokens; its exact meaning depends on the CLI counters. Input is the whole prompt, with cache reads and writes priced as in the hard head-to-head. It includes CLI context; these receipts do not separate task tokens from CLI context. It changes with cache counters. Batch timing and cache behavior were not controlled. Means, not medians, so the parts add up.",
          "polarity": "none",
          "sourceIds": [
            "calc-thinking-bill",
            "agent-provider-h2h-hard",
            "agent-effort-ladder",
            "agent-provider-h2h",
            "price-anthropic",
            "price-openai"
          ]
        },
        {
          "id": "thinking-bill-by-effort",
          "title": "Reasoning cost per strict pass by effort, with the total (calculation)",
          "subtitle": "List price ÷ strict passes; every cell is 8 tasks × 2 repetitions",
          "kind": "grouped-bar",
          "unit": "usd",
          "yLabel": "USD per strict pass (list price)",
          "series": [
            {
              "name": "Reasoning cost per strict pass",
              "points": [
                {
                  "label": "Claude Sonnet 5.5 (low) · Claude Code",
                  "value": 0.004317,
                  "n": 16
                },
                {
                  "label": "Claude Sonnet 5.5 (medium) · Claude Code",
                  "value": 0.005946,
                  "n": 16
                },
                {
                  "label": "Claude Sonnet 5.5 (high) · Claude Code",
                  "value": 0.009369,
                  "n": 16
                },
                {
                  "label": "Claude Sonnet 5.5 · Claude Code",
                  "value": 0.006299,
                  "n": 16
                },
                {
                  "label": "Claude Opus 5.5 (low) · Claude Code",
                  "value": 0.005031,
                  "n": 16
                },
                {
                  "label": "Claude Opus 5.5 (medium) · Claude Code",
                  "value": 0.01344,
                  "n": 16
                },
                {
                  "label": "Claude Opus 5.5 (high) · Claude Code",
                  "value": 0.018034,
                  "n": 16
                },
                {
                  "label": "Claude Opus 5.5 · Claude Code",
                  "value": 0.013104,
                  "n": 16
                },
                {
                  "label": "GPT-6.1 Sol (low) · Codex CLI",
                  "value": 0.001223,
                  "n": 16
                },
                {
                  "label": "GPT-6.1 Sol (medium) · Codex CLI",
                  "value": 0.002273,
                  "n": 16
                },
                {
                  "label": "GPT-6.1 Sol (high) · Codex CLI",
                  "value": 0.004114,
                  "n": 16
                }
              ]
            },
            {
              "name": "Total cost per strict pass",
              "points": [
                {
                  "label": "Claude Sonnet 5.5 (low) · Claude Code",
                  "value": 0.012191,
                  "n": 16
                },
                {
                  "label": "Claude Sonnet 5.5 (medium) · Claude Code",
                  "value": 0.01352,
                  "n": 16
                },
                {
                  "label": "Claude Sonnet 5.5 (high) · Claude Code",
                  "value": 0.016705,
                  "n": 16
                },
                {
                  "label": "Claude Sonnet 5.5 · Claude Code",
                  "value": 0.013978,
                  "n": 16
                },
                {
                  "label": "Claude Opus 5.5 (low) · Claude Code",
                  "value": 0.021152,
                  "n": 16
                },
                {
                  "label": "Claude Opus 5.5 (medium) · Claude Code",
                  "value": 0.029475,
                  "n": 16
                },
                {
                  "label": "Claude Opus 5.5 (high) · Claude Code",
                  "value": 0.033677,
                  "n": 16
                },
                {
                  "label": "Claude Opus 5.5 · Claude Code",
                  "value": 0.028925,
                  "n": 16
                },
                {
                  "label": "GPT-6.1 Sol (low) · Codex CLI",
                  "value": 0.012837,
                  "n": 16
                },
                {
                  "label": "GPT-6.1 Sol (medium) · Codex CLI",
                  "value": 0.025637,
                  "n": 16
                },
                {
                  "label": "GPT-6.1 Sol (high) · Codex CLI",
                  "value": 0.015137,
                  "n": 16
                }
              ]
            }
          ],
          "note": "Calculation, not a bill: reported tokens × list price, divided by the cell's strict passes; the calls ran on flat subscriptions. The effort-ladder cells: new calls plus reference cells reused from the hard head-to-head (Claude repetitions 1-2 only). \"Default\" means the effort flag was not passed. The total is the same value as the effort-ladder cost-per-pass chart. Effort levels are not the same scale across vendors, and the reference cells ran in a different batch and hour.",
          "polarity": "none",
          "sourceIds": [
            "calc-thinking-bill",
            "agent-provider-h2h-hard",
            "agent-effort-ladder",
            "agent-provider-h2h",
            "price-anthropic",
            "price-openai"
          ]
        },
        {
          "id": "thinking-bill-vs-time",
          "title": "Reasoning tokens vs total time per call (calculation)",
          "subtitle": "One point per call: 248 calls from the hard head-to-head and the effort ladder",
          "kind": "scatter",
          "unit": "seconds",
          "xLabel": "Reasoning tokens per call",
          "yLabel": "Total time per call (seconds)",
          "series": [
            {
              "name": "Claude Haiku 4.5 · Claude Code",
              "points": [
                {
                  "label": "Claude Haiku 4.5 · Claude Code · merge-ranges · rep 1",
                  "x": 1977,
                  "value": 18.53
                },
                {
                  "label": "Claude Haiku 4.5 · Claude Code · day-hours · rep 1",
                  "x": 4372,
                  "value": 39
                },
                {
                  "label": "Claude Haiku 4.5 · Claude Code · csv-parse · rep 1",
                  "x": 3590,
                  "value": 33.81
                },
                {
                  "label": "Claude Haiku 4.5 · Claude Code · event-loop-order · rep 1",
                  "x": 3781,
                  "value": 25.96
                },
                {
                  "label": "Claude Haiku 4.5 · Claude Code · talk-schedule · rep 1",
                  "x": 6236,
                  "value": 54.64
                },
                {
                  "label": "Claude Haiku 4.5 · Claude Code · semver-regex · rep 1",
                  "x": 7498,
                  "value": 64.48
                },
                {
                  "label": "Claude Haiku 4.5 · Claude Code · money-refactor · rep 1",
                  "x": 1452,
                  "value": 15.91
                },
                {
                  "label": "Claude Haiku 4.5 · Claude Code · q1-sql · rep 1",
                  "x": 4305,
                  "value": 38.89
                },
                {
                  "label": "Claude Haiku 4.5 · Claude Code · merge-ranges · rep 2",
                  "x": 2606,
                  "value": 24.71
                },
                {
                  "label": "Claude Haiku 4.5 · Claude Code · day-hours · rep 2",
                  "x": 3515,
                  "value": 26.15
                },
                {
                  "label": "Claude Haiku 4.5 · Claude Code · csv-parse · rep 2",
                  "x": 4497,
                  "value": 39.01
                },
                {
                  "label": "Claude Haiku 4.5 · Claude Code · event-loop-order · rep 2",
                  "x": 6575,
                  "value": 54.02
                },
                {
                  "label": "Claude Haiku 4.5 · Claude Code · talk-schedule · rep 2",
                  "x": 7954,
                  "value": 68.77
                },
                {
                  "label": "Claude Haiku 4.5 · Claude Code · semver-regex · rep 2",
                  "x": 6538,
                  "value": 54.49
                },
                {
                  "label": "Claude Haiku 4.5 · Claude Code · money-refactor · rep 2",
                  "x": 1955,
                  "value": 15.27
                },
                {
                  "label": "Claude Haiku 4.5 · Claude Code · q1-sql · rep 2",
                  "x": 6691,
                  "value": 56.13
                },
                {
                  "label": "Claude Haiku 4.5 · Claude Code · merge-ranges · rep 3",
                  "x": 2703,
                  "value": 21.22
                },
                {
                  "label": "Claude Haiku 4.5 · Claude Code · day-hours · rep 3",
                  "x": 4614,
                  "value": 40.56
                },
                {
                  "label": "Claude Haiku 4.5 · Claude Code · csv-parse · rep 3",
                  "x": 8569,
                  "value": 75.13
                },
                {
                  "label": "Claude Haiku 4.5 · Claude Code · event-loop-order · rep 3",
                  "x": 7065,
                  "value": 56.65
                },
                {
                  "label": "Claude Haiku 4.5 · Claude Code · talk-schedule · rep 3",
                  "x": 4922,
                  "value": 37.97
                },
                {
                  "label": "Claude Haiku 4.5 · Claude Code · semver-regex · rep 3",
                  "x": 7994,
                  "value": 67.03
                },
                {
                  "label": "Claude Haiku 4.5 · Claude Code · money-refactor · rep 3",
                  "x": 2221,
                  "value": 17.16
                },
                {
                  "label": "Claude Haiku 4.5 · Claude Code · q1-sql · rep 3",
                  "x": 5934,
                  "value": 47.11
                }
              ]
            },
            {
              "name": "Claude Sonnet 5.5 · Claude Code",
              "points": [
                {
                  "label": "Claude Sonnet 5.5 · Claude Code · merge-ranges · rep 1",
                  "x": 0,
                  "value": 2.93
                },
                {
                  "label": "Claude Sonnet 5.5 · Claude Code · day-hours · rep 1",
                  "x": 1249,
                  "value": 21.61
                },
                {
                  "label": "Claude Sonnet 5.5 · Claude Code · csv-parse · rep 1",
                  "x": 551,
                  "value": 8.85
                },
                {
                  "label": "Claude Sonnet 5.5 · Claude Code · event-loop-order · rep 1",
                  "x": 1064,
                  "value": 8.17
                },
                {
                  "label": "Claude Sonnet 5.5 · Claude Code · talk-schedule · rep 1",
                  "x": 716,
                  "value": 7.36
                },
                {
                  "label": "Claude Sonnet 5.5 · Claude Code · semver-regex · rep 1",
                  "x": 0,
                  "value": 2.26
                },
                {
                  "label": "Claude Sonnet 5.5 · Claude Code · money-refactor · rep 1",
                  "x": 0,
                  "value": 3.57
                },
                {
                  "label": "Claude Sonnet 5.5 · Claude Code · q1-sql · rep 1",
                  "x": 799,
                  "value": 8.91
                },
                {
                  "label": "Claude Sonnet 5.5 · Claude Code · merge-ranges · rep 2",
                  "x": 0,
                  "value": 3.06
                },
                {
                  "label": "Claude Sonnet 5.5 · Claude Code · day-hours · rep 2",
                  "x": 1614,
                  "value": 19.62
                },
                {
                  "label": "Claude Sonnet 5.5 · Claude Code · csv-parse · rep 2",
                  "x": 722,
                  "value": 9.56
                },
                {
                  "label": "Claude Sonnet 5.5 · Claude Code · event-loop-order · rep 2",
                  "x": 1197,
                  "value": 9.87
                },
                {
                  "label": "Claude Sonnet 5.5 · Claude Code · talk-schedule · rep 2",
                  "x": 736,
                  "value": 7.76
                },
                {
                  "label": "Claude Sonnet 5.5 · Claude Code · semver-regex · rep 2",
                  "x": 267,
                  "value": 3.67
                },
                {
                  "label": "Claude Sonnet 5.5 · Claude Code · money-refactor · rep 2",
                  "x": 545,
                  "value": 7.18
                },
                {
                  "label": "Claude Sonnet 5.5 · Claude Code · q1-sql · rep 2",
                  "x": 619,
                  "value": 8.47
                },
                {
                  "label": "Claude Sonnet 5.5 · Claude Code · merge-ranges · rep 3",
                  "x": 0,
                  "value": 2.38
                },
                {
                  "label": "Claude Sonnet 5.5 · Claude Code · day-hours · rep 3",
                  "x": 3060,
                  "value": 34.79
                },
                {
                  "label": "Claude Sonnet 5.5 · Claude Code · csv-parse · rep 3",
                  "x": 547,
                  "value": 10.61
                },
                {
                  "label": "Claude Sonnet 5.5 · Claude Code · event-loop-order · rep 3",
                  "x": 1177,
                  "value": 9.57
                },
                {
                  "label": "Claude Sonnet 5.5 · Claude Code · talk-schedule · rep 3",
                  "x": 657,
                  "value": 7.74
                },
                {
                  "label": "Claude Sonnet 5.5 · Claude Code · semver-regex · rep 3",
                  "x": 0,
                  "value": 2.49
                },
                {
                  "label": "Claude Sonnet 5.5 · Claude Code · money-refactor · rep 3",
                  "x": 0,
                  "value": 3.44
                },
                {
                  "label": "Claude Sonnet 5.5 · Claude Code · q1-sql · rep 3",
                  "x": 477,
                  "value": 7.56
                },
                {
                  "label": "Claude Sonnet 5.5 (low) · Claude Code · merge-ranges · rep 1",
                  "x": 0,
                  "value": 2.79
                },
                {
                  "label": "Claude Sonnet 5.5 (medium) · Claude Code · merge-ranges · rep 1",
                  "x": 0,
                  "value": 2.71
                },
                {
                  "label": "Claude Sonnet 5.5 (high) · Claude Code · merge-ranges · rep 1",
                  "x": 0,
                  "value": 2.93
                },
                {
                  "label": "Claude Sonnet 5.5 (low) · Claude Code · day-hours · rep 1",
                  "x": 1489,
                  "value": 19.96
                },
                {
                  "label": "Claude Sonnet 5.5 (medium) · Claude Code · day-hours · rep 1",
                  "x": 1941,
                  "value": 22.48
                },
                {
                  "label": "Claude Sonnet 5.5 (high) · Claude Code · day-hours · rep 1",
                  "x": 2597,
                  "value": 27.18
                },
                {
                  "label": "Claude Sonnet 5.5 (low) · Claude Code · csv-parse · rep 1",
                  "x": 0,
                  "value": 4.32
                },
                {
                  "label": "Claude Sonnet 5.5 (medium) · Claude Code · csv-parse · rep 1",
                  "x": 423,
                  "value": 7.83
                },
                {
                  "label": "Claude Sonnet 5.5 (high) · Claude Code · csv-parse · rep 1",
                  "x": 770,
                  "value": 10.01
                },
                {
                  "label": "Claude Sonnet 5.5 (low) · Claude Code · event-loop-order · rep 1",
                  "x": 995,
                  "value": 8.8
                },
                {
                  "label": "Claude Sonnet 5.5 (medium) · Claude Code · event-loop-order · rep 1",
                  "x": 1200,
                  "value": 9.98
                },
                {
                  "label": "Claude Sonnet 5.5 (high) · Claude Code · event-loop-order · rep 1",
                  "x": 1295,
                  "value": 10.89
                },
                {
                  "label": "Claude Sonnet 5.5 (low) · Claude Code · talk-schedule · rep 1",
                  "x": 598,
                  "value": 6.49
                },
                {
                  "label": "Claude Sonnet 5.5 (medium) · Claude Code · talk-schedule · rep 1",
                  "x": 651,
                  "value": 7.9
                },
                {
                  "label": "Claude Sonnet 5.5 (high) · Claude Code · talk-schedule · rep 1",
                  "x": 734,
                  "value": 9.07
                },
                {
                  "label": "Claude Sonnet 5.5 (low) · Claude Code · semver-regex · rep 1",
                  "x": 0,
                  "value": 3.51
                },
                {
                  "label": "Claude Sonnet 5.5 (medium) · Claude Code · semver-regex · rep 1",
                  "x": 251,
                  "value": 4.12
                },
                {
                  "label": "Claude Sonnet 5.5 (high) · Claude Code · semver-regex · rep 1",
                  "x": 252,
                  "value": 4.01
                },
                {
                  "label": "Claude Sonnet 5.5 (low) · Claude Code · money-refactor · rep 1",
                  "x": 0,
                  "value": 3.72
                },
                {
                  "label": "Claude Sonnet 5.5 (medium) · Claude Code · money-refactor · rep 1",
                  "x": 0,
                  "value": 3.89
                },
                {
                  "label": "Claude Sonnet 5.5 (high) · Claude Code · money-refactor · rep 1",
                  "x": 693,
                  "value": 8.09
                },
                {
                  "label": "Claude Sonnet 5.5 (low) · Claude Code · q1-sql · rep 1",
                  "x": 545,
                  "value": 7.92
                },
                {
                  "label": "Claude Sonnet 5.5 (medium) · Claude Code · q1-sql · rep 1",
                  "x": 421,
                  "value": 7.43
                },
                {
                  "label": "Claude Sonnet 5.5 (high) · Claude Code · q1-sql · rep 1",
                  "x": 826,
                  "value": 10.79
                },
                {
                  "label": "Claude Sonnet 5.5 (low) · Claude Code · merge-ranges · rep 2",
                  "x": 0,
                  "value": 2.78
                },
                {
                  "label": "Claude Sonnet 5.5 (medium) · Claude Code · merge-ranges · rep 2",
                  "x": 0,
                  "value": 2.93
                },
                {
                  "label": "Claude Sonnet 5.5 (high) · Claude Code · merge-ranges · rep 2",
                  "x": 0,
                  "value": 3.49
                },
                {
                  "label": "Claude Sonnet 5.5 (low) · Claude Code · day-hours · rep 2",
                  "x": 1091,
                  "value": 16.28
                },
                {
                  "label": "Claude Sonnet 5.5 (medium) · Claude Code · day-hours · rep 2",
                  "x": 2093,
                  "value": 24.01
                },
                {
                  "label": "Claude Sonnet 5.5 (high) · Claude Code · day-hours · rep 2",
                  "x": 3610,
                  "value": 35.81
                },
                {
                  "label": "Claude Sonnet 5.5 (low) · Claude Code · csv-parse · rep 2",
                  "x": 0,
                  "value": 4.4
                },
                {
                  "label": "Claude Sonnet 5.5 (medium) · Claude Code · csv-parse · rep 2",
                  "x": 0,
                  "value": 4.65
                },
                {
                  "label": "Claude Sonnet 5.5 (high) · Claude Code · csv-parse · rep 2",
                  "x": 783,
                  "value": 16.1
                },
                {
                  "label": "Claude Sonnet 5.5 (low) · Claude Code · event-loop-order · rep 2",
                  "x": 974,
                  "value": 9.73
                },
                {
                  "label": "Claude Sonnet 5.5 (medium) · Claude Code · event-loop-order · rep 2",
                  "x": 1098,
                  "value": 8.72
                },
                {
                  "label": "Claude Sonnet 5.5 (high) · Claude Code · event-loop-order · rep 2",
                  "x": 1226,
                  "value": 10.77
                },
                {
                  "label": "Claude Sonnet 5.5 (low) · Claude Code · talk-schedule · rep 2",
                  "x": 624,
                  "value": 6.71
                },
                {
                  "label": "Claude Sonnet 5.5 (medium) · Claude Code · talk-schedule · rep 2",
                  "x": 624,
                  "value": 9.61
                },
                {
                  "label": "Claude Sonnet 5.5 (high) · Claude Code · talk-schedule · rep 2",
                  "x": 755,
                  "value": 8.1
                },
                {
                  "label": "Claude Sonnet 5.5 (low) · Claude Code · semver-regex · rep 2",
                  "x": 0,
                  "value": 2.86
                },
                {
                  "label": "Claude Sonnet 5.5 (medium) · Claude Code · semver-regex · rep 2",
                  "x": 245,
                  "value": 3.82
                },
                {
                  "label": "Claude Sonnet 5.5 (high) · Claude Code · semver-regex · rep 2",
                  "x": 243,
                  "value": 3.71
                },
                {
                  "label": "Claude Sonnet 5.5 (low) · Claude Code · money-refactor · rep 2",
                  "x": 0,
                  "value": 5.16
                },
                {
                  "label": "Claude Sonnet 5.5 (medium) · Claude Code · money-refactor · rep 2",
                  "x": 0,
                  "value": 3.99
                },
                {
                  "label": "Claude Sonnet 5.5 (high) · Claude Code · money-refactor · rep 2",
                  "x": 580,
                  "value": 7.61
                },
                {
                  "label": "Claude Sonnet 5.5 (low) · Claude Code · q1-sql · rep 2",
                  "x": 591,
                  "value": 7.68
                },
                {
                  "label": "Claude Sonnet 5.5 (medium) · Claude Code · q1-sql · rep 2",
                  "x": 566,
                  "value": 9.18
                },
                {
                  "label": "Claude Sonnet 5.5 (high) · Claude Code · q1-sql · rep 2",
                  "x": 626,
                  "value": 8.55
                }
              ]
            },
            {
              "name": "Claude Opus 5.5 · Claude Code",
              "points": [
                {
                  "label": "Claude Opus 5.5 · Claude Code · merge-ranges · rep 1",
                  "x": 100,
                  "value": 4.24
                },
                {
                  "label": "Claude Opus 5.5 (high) · Claude Code · merge-ranges · rep 1",
                  "x": 188,
                  "value": 5.36
                },
                {
                  "label": "Claude Opus 5.5 · Claude Code · day-hours · rep 1",
                  "x": 1580,
                  "value": 26.3
                },
                {
                  "label": "Claude Opus 5.5 (high) · Claude Code · day-hours · rep 1",
                  "x": 3301,
                  "value": 63
                },
                {
                  "label": "Claude Opus 5.5 · Claude Code · csv-parse · rep 1",
                  "x": 543,
                  "value": 10.98
                },
                {
                  "label": "Claude Opus 5.5 (high) · Claude Code · csv-parse · rep 1",
                  "x": 528,
                  "value": 11.1
                },
                {
                  "label": "Claude Opus 5.5 · Claude Code · event-loop-order · rep 1",
                  "x": 958,
                  "value": 11.91
                },
                {
                  "label": "Claude Opus 5.5 (high) · Claude Code · event-loop-order · rep 1",
                  "x": 1274,
                  "value": 13.6
                },
                {
                  "label": "Claude Opus 5.5 · Claude Code · talk-schedule · rep 1",
                  "x": 574,
                  "value": 7.82
                },
                {
                  "label": "Claude Opus 5.5 (high) · Claude Code · talk-schedule · rep 1",
                  "x": 656,
                  "value": 8.9
                },
                {
                  "label": "Claude Opus 5.5 · Claude Code · semver-regex · rep 1",
                  "x": 300,
                  "value": 5.35
                },
                {
                  "label": "Claude Opus 5.5 (high) · Claude Code · semver-regex · rep 1",
                  "x": 293,
                  "value": 5
                },
                {
                  "label": "Claude Opus 5.5 · Claude Code · money-refactor · rep 1",
                  "x": 429,
                  "value": 8.15
                },
                {
                  "label": "Claude Opus 5.5 (high) · Claude Code · money-refactor · rep 1",
                  "x": 528,
                  "value": 9.12
                },
                {
                  "label": "Claude Opus 5.5 · Claude Code · q1-sql · rep 1",
                  "x": 796,
                  "value": 12.78
                },
                {
                  "label": "Claude Opus 5.5 (high) · Claude Code · q1-sql · rep 1",
                  "x": 920,
                  "value": 13.66
                },
                {
                  "label": "Claude Opus 5.5 · Claude Code · merge-ranges · rep 2",
                  "x": 114,
                  "value": 4.65
                },
                {
                  "label": "Claude Opus 5.5 (high) · Claude Code · merge-ranges · rep 2",
                  "x": 182,
                  "value": 5.52
                },
                {
                  "label": "Claude Opus 5.5 · Claude Code · day-hours · rep 2",
                  "x": 1778,
                  "value": 27.21
                },
                {
                  "label": "Claude Opus 5.5 (high) · Claude Code · day-hours · rep 2",
                  "x": 2756,
                  "value": 36.44
                },
                {
                  "label": "Claude Opus 5.5 · Claude Code · csv-parse · rep 2",
                  "x": 524,
                  "value": 11.47
                },
                {
                  "label": "Claude Opus 5.5 (high) · Claude Code · csv-parse · rep 2",
                  "x": 573,
                  "value": 12.22
                },
                {
                  "label": "Claude Opus 5.5 · Claude Code · event-loop-order · rep 2",
                  "x": 1015,
                  "value": 10.2
                },
                {
                  "label": "Claude Opus 5.5 (high) · Claude Code · event-loop-order · rep 2",
                  "x": 1300,
                  "value": 13.45
                },
                {
                  "label": "Claude Opus 5.5 · Claude Code · talk-schedule · rep 2",
                  "x": 533,
                  "value": 7.63
                },
                {
                  "label": "Claude Opus 5.5 (high) · Claude Code · talk-schedule · rep 2",
                  "x": 655,
                  "value": 8.54
                },
                {
                  "label": "Claude Opus 5.5 · Claude Code · semver-regex · rep 2",
                  "x": 244,
                  "value": 4.75
                },
                {
                  "label": "Claude Opus 5.5 (high) · Claude Code · semver-regex · rep 2",
                  "x": 103,
                  "value": 3.63
                },
                {
                  "label": "Claude Opus 5.5 · Claude Code · money-refactor · rep 2",
                  "x": 214,
                  "value": 7.2
                },
                {
                  "label": "Claude Opus 5.5 (high) · Claude Code · money-refactor · rep 2",
                  "x": 431,
                  "value": 8.84
                },
                {
                  "label": "Claude Opus 5.5 · Claude Code · q1-sql · rep 2",
                  "x": 781,
                  "value": 12.44
                },
                {
                  "label": "Claude Opus 5.5 (high) · Claude Code · q1-sql · rep 2",
                  "x": 739,
                  "value": 12.46
                },
                {
                  "label": "Claude Opus 5.5 · Claude Code · merge-ranges · rep 3",
                  "x": 122,
                  "value": 4.45
                },
                {
                  "label": "Claude Opus 5.5 (high) · Claude Code · merge-ranges · rep 3",
                  "x": 221,
                  "value": 5.73
                },
                {
                  "label": "Claude Opus 5.5 · Claude Code · day-hours · rep 3",
                  "x": 1136,
                  "value": 19.74
                },
                {
                  "label": "Claude Opus 5.5 (high) · Claude Code · day-hours · rep 3",
                  "x": 3027,
                  "value": 38.79
                },
                {
                  "label": "Claude Opus 5.5 · Claude Code · csv-parse · rep 3",
                  "x": 440,
                  "value": 11.67
                },
                {
                  "label": "Claude Opus 5.5 (high) · Claude Code · csv-parse · rep 3",
                  "x": 800,
                  "value": 12.62
                },
                {
                  "label": "Claude Opus 5.5 · Claude Code · event-loop-order · rep 3",
                  "x": 1109,
                  "value": 11.48
                },
                {
                  "label": "Claude Opus 5.5 (high) · Claude Code · event-loop-order · rep 3",
                  "x": 1198,
                  "value": 13.5
                },
                {
                  "label": "Claude Opus 5.5 · Claude Code · talk-schedule · rep 3",
                  "x": 444,
                  "value": 6.88
                },
                {
                  "label": "Claude Opus 5.5 (high) · Claude Code · talk-schedule · rep 3",
                  "x": 553,
                  "value": 7.46
                },
                {
                  "label": "Claude Opus 5.5 · Claude Code · semver-regex · rep 3",
                  "x": 293,
                  "value": 5.06
                },
                {
                  "label": "Claude Opus 5.5 (high) · Claude Code · semver-regex · rep 3",
                  "x": 122,
                  "value": 3.98
                },
                {
                  "label": "Claude Opus 5.5 · Claude Code · money-refactor · rep 3",
                  "x": 199,
                  "value": 6.19
                },
                {
                  "label": "Claude Opus 5.5 (high) · Claude Code · money-refactor · rep 3",
                  "x": 304,
                  "value": 10.95
                },
                {
                  "label": "Claude Opus 5.5 · Claude Code · q1-sql · rep 3",
                  "x": 808,
                  "value": 12.96
                },
                {
                  "label": "Claude Opus 5.5 (high) · Claude Code · q1-sql · rep 3",
                  "x": 911,
                  "value": 12.91
                },
                {
                  "label": "Claude Opus 5.5 (low) · Claude Code · merge-ranges · rep 1",
                  "x": 0,
                  "value": 3.34
                },
                {
                  "label": "Claude Opus 5.5 (medium) · Claude Code · merge-ranges · rep 1",
                  "x": 109,
                  "value": 5.51
                },
                {
                  "label": "Claude Opus 5.5 (low) · Claude Code · day-hours · rep 1",
                  "x": 394,
                  "value": 13.02
                },
                {
                  "label": "Claude Opus 5.5 (medium) · Claude Code · day-hours · rep 1",
                  "x": 1993,
                  "value": 31.12
                },
                {
                  "label": "Claude Opus 5.5 (low) · Claude Code · csv-parse · rep 1",
                  "x": 471,
                  "value": 9.88
                },
                {
                  "label": "Claude Opus 5.5 (medium) · Claude Code · csv-parse · rep 1",
                  "x": 444,
                  "value": 18.26
                },
                {
                  "label": "Claude Opus 5.5 (low) · Claude Code · event-loop-order · rep 1",
                  "x": 684,
                  "value": 8.7
                },
                {
                  "label": "Claude Opus 5.5 (medium) · Claude Code · event-loop-order · rep 1",
                  "x": 1120,
                  "value": 12.42
                },
                {
                  "label": "Claude Opus 5.5 (low) · Claude Code · talk-schedule · rep 1",
                  "x": 498,
                  "value": 7.72
                },
                {
                  "label": "Claude Opus 5.5 (medium) · Claude Code · talk-schedule · rep 1",
                  "x": 471,
                  "value": 6.99
                },
                {
                  "label": "Claude Opus 5.5 (low) · Claude Code · semver-regex · rep 1",
                  "x": 88,
                  "value": 4.21
                },
                {
                  "label": "Claude Opus 5.5 (medium) · Claude Code · semver-regex · rep 1",
                  "x": 119,
                  "value": 6.8
                },
                {
                  "label": "Claude Opus 5.5 (low) · Claude Code · money-refactor · rep 1",
                  "x": 0,
                  "value": 5.65
                },
                {
                  "label": "Claude Opus 5.5 (medium) · Claude Code · money-refactor · rep 1",
                  "x": 183,
                  "value": 6.87
                },
                {
                  "label": "Claude Opus 5.5 (low) · Claude Code · q1-sql · rep 1",
                  "x": 0,
                  "value": 10.64
                },
                {
                  "label": "Claude Opus 5.5 (medium) · Claude Code · q1-sql · rep 1",
                  "x": 793,
                  "value": 13.13
                },
                {
                  "label": "Claude Opus 5.5 (low) · Claude Code · merge-ranges · rep 2",
                  "x": 0,
                  "value": 3.44
                },
                {
                  "label": "Claude Opus 5.5 (medium) · Claude Code · merge-ranges · rep 2",
                  "x": 115,
                  "value": 5.29
                },
                {
                  "label": "Claude Opus 5.5 (low) · Claude Code · day-hours · rep 2",
                  "x": 587,
                  "value": 15.82
                },
                {
                  "label": "Claude Opus 5.5 (medium) · Claude Code · day-hours · rep 2",
                  "x": 1846,
                  "value": 31.36
                },
                {
                  "label": "Claude Opus 5.5 (low) · Claude Code · csv-parse · rep 2",
                  "x": 0,
                  "value": 6.21
                },
                {
                  "label": "Claude Opus 5.5 (medium) · Claude Code · csv-parse · rep 2",
                  "x": 655,
                  "value": 10.91
                },
                {
                  "label": "Claude Opus 5.5 (low) · Claude Code · event-loop-order · rep 2",
                  "x": 791,
                  "value": 9.16
                },
                {
                  "label": "Claude Opus 5.5 (medium) · Claude Code · event-loop-order · rep 2",
                  "x": 1014,
                  "value": 11.19
                },
                {
                  "label": "Claude Opus 5.5 (low) · Claude Code · talk-schedule · rep 2",
                  "x": 426,
                  "value": 7.29
                },
                {
                  "label": "Claude Opus 5.5 (medium) · Claude Code · talk-schedule · rep 2",
                  "x": 565,
                  "value": 8.53
                },
                {
                  "label": "Claude Opus 5.5 (low) · Claude Code · semver-regex · rep 2",
                  "x": 86,
                  "value": 3.79
                },
                {
                  "label": "Claude Opus 5.5 (medium) · Claude Code · semver-regex · rep 2",
                  "x": 243,
                  "value": 4.78
                },
                {
                  "label": "Claude Opus 5.5 (low) · Claude Code · money-refactor · rep 2",
                  "x": 0,
                  "value": 4.93
                },
                {
                  "label": "Claude Opus 5.5 (medium) · Claude Code · money-refactor · rep 2",
                  "x": 253,
                  "value": 7.07
                },
                {
                  "label": "Claude Opus 5.5 (low) · Claude Code · q1-sql · rep 2",
                  "x": 0,
                  "value": 9.23
                },
                {
                  "label": "Claude Opus 5.5 (medium) · Claude Code · q1-sql · rep 2",
                  "x": 829,
                  "value": 13.8
                }
              ]
            },
            {
              "name": "Claude Fable 5.1 · Claude Code",
              "points": [
                {
                  "label": "Claude Fable 5.1 · Claude Code · merge-ranges · rep 1",
                  "x": 85,
                  "value": 4.76
                },
                {
                  "label": "Claude Fable 5.1 · Claude Code · day-hours · rep 1",
                  "x": 1815,
                  "value": 33.71
                },
                {
                  "label": "Claude Fable 5.1 · Claude Code · csv-parse · rep 1",
                  "x": 903,
                  "value": 16.2
                },
                {
                  "label": "Claude Fable 5.1 · Claude Code · event-loop-order · rep 1",
                  "x": 1765,
                  "value": 23.9
                },
                {
                  "label": "Claude Fable 5.1 · Claude Code · talk-schedule · rep 1",
                  "x": 1090,
                  "value": 17.27
                },
                {
                  "label": "Claude Fable 5.1 · Claude Code · semver-regex · rep 1",
                  "x": 261,
                  "value": 9.6
                },
                {
                  "label": "Claude Fable 5.1 · Claude Code · money-refactor · rep 1",
                  "x": 646,
                  "value": 12.91
                },
                {
                  "label": "Claude Fable 5.1 · Claude Code · q1-sql · rep 1",
                  "x": 875,
                  "value": 16.07
                },
                {
                  "label": "Claude Fable 5.1 · Claude Code · merge-ranges · rep 2",
                  "x": 75,
                  "value": 4.46
                },
                {
                  "label": "Claude Fable 5.1 · Claude Code · day-hours · rep 2",
                  "x": 5889,
                  "value": 90
                },
                {
                  "label": "Claude Fable 5.1 · Claude Code · csv-parse · rep 2",
                  "x": 1088,
                  "value": 21.38
                },
                {
                  "label": "Claude Fable 5.1 · Claude Code · event-loop-order · rep 2",
                  "x": 1228,
                  "value": 16.78
                },
                {
                  "label": "Claude Fable 5.1 · Claude Code · talk-schedule · rep 2",
                  "x": 786,
                  "value": 11.23
                },
                {
                  "label": "Claude Fable 5.1 · Claude Code · semver-regex · rep 2",
                  "x": 236,
                  "value": 7.52
                },
                {
                  "label": "Claude Fable 5.1 · Claude Code · money-refactor · rep 2",
                  "x": 796,
                  "value": 13.68
                },
                {
                  "label": "Claude Fable 5.1 · Claude Code · q1-sql · rep 2",
                  "x": 865,
                  "value": 20.31
                },
                {
                  "label": "Claude Fable 5.1 · Claude Code · merge-ranges · rep 3",
                  "x": 197,
                  "value": 15.21
                },
                {
                  "label": "Claude Fable 5.1 · Claude Code · day-hours · rep 3",
                  "x": 1427,
                  "value": 25.09
                },
                {
                  "label": "Claude Fable 5.1 · Claude Code · csv-parse · rep 3",
                  "x": 1000,
                  "value": 15.88
                },
                {
                  "label": "Claude Fable 5.1 · Claude Code · event-loop-order · rep 3",
                  "x": 1606,
                  "value": 26.91
                },
                {
                  "label": "Claude Fable 5.1 · Claude Code · talk-schedule · rep 3",
                  "x": 836,
                  "value": 12.14
                },
                {
                  "label": "Claude Fable 5.1 · Claude Code · semver-regex · rep 3",
                  "x": 266,
                  "value": 5.72
                },
                {
                  "label": "Claude Fable 5.1 · Claude Code · money-refactor · rep 3",
                  "x": 948,
                  "value": 21.85
                },
                {
                  "label": "Claude Fable 5.1 · Claude Code · q1-sql · rep 3",
                  "x": 1091,
                  "value": 24.58
                }
              ]
            },
            {
              "name": "GPT-6.1 Sol · Codex CLI",
              "points": [
                {
                  "label": "GPT-6.1 Sol (medium) · Codex CLI · merge-ranges · rep 1",
                  "x": 163,
                  "value": 13.2
                },
                {
                  "label": "GPT-6.1 Sol (high) · Codex CLI · merge-ranges · rep 1",
                  "x": 192,
                  "value": 14.4
                },
                {
                  "label": "GPT-6.1 Sol (medium) · Codex CLI · day-hours · rep 1",
                  "x": 830,
                  "value": 61.6
                },
                {
                  "label": "GPT-6.1 Sol (high) · Codex CLI · day-hours · rep 1",
                  "x": 1533,
                  "value": 82.99
                },
                {
                  "label": "GPT-6.1 Sol (medium) · Codex CLI · csv-parse · rep 1",
                  "x": 86,
                  "value": 14.79
                },
                {
                  "label": "GPT-6.1 Sol (high) · Codex CLI · csv-parse · rep 1",
                  "x": 256,
                  "value": 22.91
                },
                {
                  "label": "GPT-6.1 Sol (medium) · Codex CLI · event-loop-order · rep 1",
                  "x": 199,
                  "value": 8.54
                },
                {
                  "label": "GPT-6.1 Sol (high) · Codex CLI · event-loop-order · rep 1",
                  "x": 347,
                  "value": 15.44
                },
                {
                  "label": "GPT-6.1 Sol (medium) · Codex CLI · talk-schedule · rep 1",
                  "x": 189,
                  "value": 12.63
                },
                {
                  "label": "GPT-6.1 Sol (high) · Codex CLI · talk-schedule · rep 1",
                  "x": 189,
                  "value": 13.05
                },
                {
                  "label": "GPT-6.1 Sol (medium) · Codex CLI · semver-regex · rep 1",
                  "x": 101,
                  "value": 9.63
                },
                {
                  "label": "GPT-6.1 Sol (high) · Codex CLI · semver-regex · rep 1",
                  "x": 306,
                  "value": 17.84
                },
                {
                  "label": "GPT-6.1 Sol (medium) · Codex CLI · money-refactor · rep 1",
                  "x": 89,
                  "value": 11.74
                },
                {
                  "label": "GPT-6.1 Sol (high) · Codex CLI · money-refactor · rep 1",
                  "x": 224,
                  "value": 19.39
                },
                {
                  "label": "GPT-6.1 Sol (medium) · Codex CLI · q1-sql · rep 1",
                  "x": 61,
                  "value": 13.76
                },
                {
                  "label": "GPT-6.1 Sol (high) · Codex CLI · q1-sql · rep 1",
                  "x": 181,
                  "value": 18.4
                },
                {
                  "label": "GPT-6.1 Sol (medium) · Codex CLI · merge-ranges · rep 2",
                  "x": 137,
                  "value": 14.63
                },
                {
                  "label": "GPT-6.1 Sol (high) · Codex CLI · merge-ranges · rep 2",
                  "x": 226,
                  "value": 15.12
                },
                {
                  "label": "GPT-6.1 Sol (medium) · Codex CLI · day-hours · rep 2",
                  "x": 839,
                  "value": 56.39
                },
                {
                  "label": "GPT-6.1 Sol (high) · Codex CLI · day-hours · rep 2",
                  "x": 1750,
                  "value": 92.21
                },
                {
                  "label": "GPT-6.1 Sol (medium) · Codex CLI · csv-parse · rep 2",
                  "x": 112,
                  "value": 16.64
                },
                {
                  "label": "GPT-6.1 Sol (high) · Codex CLI · csv-parse · rep 2",
                  "x": 220,
                  "value": 20.56
                },
                {
                  "label": "GPT-6.1 Sol (medium) · Codex CLI · event-loop-order · rep 2",
                  "x": 251,
                  "value": 12.5
                },
                {
                  "label": "GPT-6.1 Sol (high) · Codex CLI · event-loop-order · rep 2",
                  "x": 375,
                  "value": 22.51
                },
                {
                  "label": "GPT-6.1 Sol (medium) · Codex CLI · talk-schedule · rep 2",
                  "x": 197,
                  "value": 11.81
                },
                {
                  "label": "GPT-6.1 Sol (high) · Codex CLI · talk-schedule · rep 2",
                  "x": 229,
                  "value": 12.21
                },
                {
                  "label": "GPT-6.1 Sol (medium) · Codex CLI · semver-regex · rep 2",
                  "x": 193,
                  "value": 13.01
                },
                {
                  "label": "GPT-6.1 Sol (high) · Codex CLI · semver-regex · rep 2",
                  "x": 189,
                  "value": 11.67
                },
                {
                  "label": "GPT-6.1 Sol (medium) · Codex CLI · money-refactor · rep 2",
                  "x": 71,
                  "value": 11.85
                },
                {
                  "label": "GPT-6.1 Sol (high) · Codex CLI · money-refactor · rep 2",
                  "x": 144,
                  "value": 15.43
                },
                {
                  "label": "GPT-6.1 Sol (medium) · Codex CLI · q1-sql · rep 2",
                  "x": 119,
                  "value": 17.81
                },
                {
                  "label": "GPT-6.1 Sol (high) · Codex CLI · q1-sql · rep 2",
                  "x": 222,
                  "value": 19.62
                },
                {
                  "label": "GPT-6.1 Sol (low) · Codex CLI · merge-ranges · rep 1",
                  "x": 63,
                  "value": 17.2
                },
                {
                  "label": "GPT-6.1 Sol (low) · Codex CLI · day-hours · rep 1",
                  "x": 458,
                  "value": 44.29
                },
                {
                  "label": "GPT-6.1 Sol (low) · Codex CLI · csv-parse · rep 1",
                  "x": 78,
                  "value": 17.66
                },
                {
                  "label": "GPT-6.1 Sol (low) · Codex CLI · event-loop-order · rep 1",
                  "x": 198,
                  "value": 14.61
                },
                {
                  "label": "GPT-6.1 Sol (low) · Codex CLI · talk-schedule · rep 1",
                  "x": 189,
                  "value": 13.6
                },
                {
                  "label": "GPT-6.1 Sol (low) · Codex CLI · semver-regex · rep 1",
                  "x": 44,
                  "value": 8.57
                },
                {
                  "label": "GPT-6.1 Sol (low) · Codex CLI · money-refactor · rep 1",
                  "x": 62,
                  "value": 13.12
                },
                {
                  "label": "GPT-6.1 Sol (low) · Codex CLI · q1-sql · rep 1",
                  "x": 0,
                  "value": 13.64
                },
                {
                  "label": "GPT-6.1 Sol (low) · Codex CLI · merge-ranges · rep 2",
                  "x": 0,
                  "value": 7.94
                },
                {
                  "label": "GPT-6.1 Sol (low) · Codex CLI · day-hours · rep 2",
                  "x": 400,
                  "value": 40.71
                },
                {
                  "label": "GPT-6.1 Sol (low) · Codex CLI · csv-parse · rep 2",
                  "x": 0,
                  "value": 17.85
                },
                {
                  "label": "GPT-6.1 Sol (low) · Codex CLI · event-loop-order · rep 2",
                  "x": 185,
                  "value": 11.1
                },
                {
                  "label": "GPT-6.1 Sol (low) · Codex CLI · talk-schedule · rep 2",
                  "x": 189,
                  "value": 12.41
                },
                {
                  "label": "GPT-6.1 Sol (low) · Codex CLI · semver-regex · rep 2",
                  "x": 50,
                  "value": 8.9
                },
                {
                  "label": "GPT-6.1 Sol (low) · Codex CLI · money-refactor · rep 2",
                  "x": 0,
                  "value": 9.54
                },
                {
                  "label": "GPT-6.1 Sol (low) · Codex CLI · q1-sql · rep 2",
                  "x": 40,
                  "value": 14.42
                }
              ]
            }
          ],
          "note": "Calculation, not a run: each point is one recorded call. Spearman rank correlations and slopes are in the table. Total time includes CLI start-up and the visible answer. Task, effort and batch can affect both counts and time. The plot shows association, not cause. 248 calls over 8 tasks; calls are not independent.",
          "polarity": "none",
          "sourceIds": [
            "calc-thinking-bill",
            "agent-provider-h2h-hard",
            "agent-effort-ladder",
            "agent-provider-h2h",
            "price-anthropic",
            "price-openai"
          ]
        },
        {
          "id": "thinking-bill-short-vs-hard",
          "title": "Reasoning share on short tasks vs hard tasks (calculation)",
          "subtitle": "Median call per configuration; five short tasks and eight hard tasks",
          "kind": "grouped-bar",
          "unit": "percent",
          "yLabel": "Reasoning share of output tokens (median call, %)",
          "series": [
            {
              "name": "Eight hard tasks",
              "points": [
                {
                  "label": "Claude Haiku 4.5 · Claude Code",
                  "value": 91.68,
                  "lo": 76.46,
                  "hi": 99.27,
                  "n": 24
                },
                {
                  "label": "Claude Sonnet 5.5 · Claude Code",
                  "value": 54.54,
                  "lo": 0,
                  "hi": 95.91,
                  "n": 24
                },
                {
                  "label": "Claude Opus 5.5 · Claude Code",
                  "value": 54.79,
                  "lo": 29.92,
                  "hi": 95.6,
                  "n": 24
                },
                {
                  "label": "Claude Opus 5.5 (high) · Claude Code",
                  "value": 54.43,
                  "lo": 36.14,
                  "hi": 96.23,
                  "n": 24
                },
                {
                  "label": "Claude Fable 5.1 · Claude Code",
                  "value": 64.24,
                  "lo": 23.44,
                  "hi": 97.19,
                  "n": 24
                },
                {
                  "label": "GPT-6.1 Sol (medium) · Codex CLI",
                  "value": 46.33,
                  "lo": 11.42,
                  "hi": 86.85,
                  "n": 16
                },
                {
                  "label": "GPT-6.1 Sol (high) · Codex CLI",
                  "value": 57.01,
                  "lo": 29.19,
                  "hi": 90.8,
                  "n": 16
                }
              ]
            },
            {
              "name": "Five short tasks",
              "points": [
                {
                  "label": "Claude Haiku 4.5 · Claude Code",
                  "value": 90.19,
                  "lo": 73.1,
                  "hi": 97.59,
                  "n": 15
                },
                {
                  "label": "Claude Sonnet 5.5 · Claude Code",
                  "value": 0,
                  "lo": 0,
                  "hi": 72.75,
                  "n": 15
                },
                {
                  "label": "Claude Opus 5.5 · Claude Code",
                  "value": 0,
                  "lo": 0,
                  "hi": 93.33,
                  "n": 15
                },
                {
                  "label": "Claude Opus 5.5 (high) · Claude Code",
                  "value": 43.59,
                  "lo": 0,
                  "hi": 93.33,
                  "n": 15
                },
                {
                  "label": "Claude Fable 5.1 · Claude Code",
                  "value": 0,
                  "lo": 0,
                  "hi": 74.01,
                  "n": 15
                },
                {
                  "label": "GPT-6.1 Sol (medium) · Codex CLI",
                  "value": 41.05,
                  "lo": 0,
                  "hi": 71.43,
                  "n": 15
                },
                {
                  "label": "GPT-6.1 Sol (high) · Codex CLI",
                  "value": 58.06,
                  "lo": 0,
                  "hi": 75.76,
                  "n": 15
                }
              ]
            }
          ],
          "note": "Calculation from reported tokens, not a run: the median of per-call reasoning ÷ output. A median of 0% means at least half of the calls reported 0 reasoning tokens in the receipt. On a route that reports reasoning, we retain a numeric 0 as the recorded counter. It does not prove the model did no internal reasoning. The table says how many calls reported 0. The short tasks are five small validated tasks (a bug fix, a JSON extraction, an arithmetic problem, a refactor and a ticket classification). Per-call ranges are in the tables.",
          "whisker": "minmax",
          "polarity": "none",
          "sourceIds": [
            "calc-thinking-bill",
            "agent-provider-h2h-hard",
            "agent-effort-ladder",
            "agent-provider-h2h",
            "price-anthropic",
            "price-openai"
          ]
        }
      ],
      "tables": [
        {
          "id": "thinking-bill-hard-cells",
          "title": "Reasoning tokens and list-price cost per call, hard tasks (calculation)",
          "columns": [
            {
              "key": "config",
              "label": "Configuration",
              "unit": "text"
            },
            {
              "key": "calls",
              "label": "Calls",
              "unit": "count"
            },
            {
              "key": "strict",
              "label": "Strict passes",
              "unit": "text"
            },
            {
              "key": "medOut",
              "label": "Median output tokens",
              "unit": "tokens"
            },
            {
              "key": "medRea",
              "label": "Median reasoning tokens",
              "unit": "tokens"
            },
            {
              "key": "tokenRange",
              "label": "Output / reasoning tokens, lowest to highest call",
              "unit": "text"
            },
            {
              "key": "costRange",
              "label": "Reasoning / total cost, lowest to highest call (USD, calculation)",
              "unit": "text"
            },
            {
              "key": "costShareRange",
              "label": "Reasoning cost share, lowest to highest call (calculation)",
              "unit": "text"
            },
            {
              "key": "zero",
              "label": "Calls with 0 reasoning tokens",
              "unit": "count"
            },
            {
              "key": "medShare",
              "label": "Median call: reasoning share of output",
              "unit": "rate"
            },
            {
              "key": "shareRange",
              "label": "Share, lowest to highest call",
              "unit": "text"
            },
            {
              "key": "pooled",
              "label": "Pooled share (all reasoning ÷ all output)",
              "unit": "rate"
            },
            {
              "key": "meanReasoning",
              "label": "Reasoning cost per call (USD, calculation)",
              "unit": "usd"
            },
            {
              "key": "meanVisible",
              "label": "Remaining output cost per call (USD, calculation)",
              "unit": "usd"
            },
            {
              "key": "meanInput",
              "label": "Input cost per call (USD)",
              "unit": "usd"
            },
            {
              "key": "meanTotal",
              "label": "Total cost per call (USD)",
              "unit": "usd"
            },
            {
              "key": "costShare",
              "label": "Pooled reasoning share of list-price cost (calculation)",
              "unit": "rate"
            },
            {
              "key": "reasoningPerPass",
              "label": "Reasoning cost per strict pass (USD)",
              "unit": "usd"
            },
            {
              "key": "totalPerPass",
              "label": "Total cost per strict pass (USD)",
              "unit": "usd"
            }
          ],
          "rows": [
            {
              "config": "Claude Haiku 4.5 · Claude Code",
              "calls": 24,
              "strict": "11/24 (95% Wilson 28% to 65%)",
              "medOut": 5064,
              "medRea": 4556,
              "tokenRange": "1899 to 9321 / 1452 to 8569",
              "costRange": "$0.0073 to $0.0428 / $0.0148 to $0.0505",
              "costShareRange": "40% to 90%",
              "zero": 0,
              "medShare": 0.9168,
              "shareRange": "76% to 99%",
              "pooled": 0.9317,
              "meanReasoning": 0.024492,
              "meanVisible": 0.001795,
              "meanInput": 0.00451,
              "meanTotal": 0.030798,
              "costShare": 0.7953,
              "reasoningPerPass": 0.053438,
              "totalPerPass": 0.067196
            },
            {
              "config": "Claude Sonnet 5.5 · Claude Code",
              "calls": 24,
              "strict": "24/24 (95% Wilson 86% to 100%)",
              "medOut": 1050,
              "medRea": 585,
              "tokenRange": "176 to 3895 / 0 to 3060",
              "costRange": "$0.0000 to $0.0306 / $0.0051 to $0.0424",
              "costShareRange": "0% to 74%",
              "zero": 7,
              "medShare": 0.5454,
              "shareRange": "0% to 96%",
              "pooled": 0.6448,
              "meanReasoning": 0.006665,
              "meanVisible": 0.003672,
              "meanInput": 0.004012,
              "meanTotal": 0.014349,
              "costShare": 0.4645,
              "reasoningPerPass": 0.006665,
              "totalPerPass": 0.014349
            },
            {
              "config": "Claude Opus 5.5 · Claude Code",
              "calls": 24,
              "strict": "24/24 (95% Wilson 86% to 100%)",
              "medOut": 945,
              "medRea": 529,
              "tokenRange": "323 to 2531 / 100 to 1778",
              "costRange": "$0.0020 to $0.0356 / $0.0131 to $0.0572",
              "costShareRange": "10% to 74%",
              "zero": 0,
              "medShare": 0.5479,
              "shareRange": "30% to 96%",
              "pooled": 0.6102,
              "meanReasoning": 0.012528,
              "meanVisible": 0.008003,
              "meanInput": 0.00771,
              "meanTotal": 0.028241,
              "costShare": 0.4436,
              "reasoningPerPass": 0.012528,
              "totalPerPass": 0.028241
            },
            {
              "config": "Claude Opus 5.5 (high) · Claude Code",
              "calls": 24,
              "strict": "24/24 (95% Wilson 86% to 100%)",
              "medOut": 1052,
              "medRea": 614,
              "tokenRange": "285 to 4052 / 103 to 3301",
              "costRange": "$0.0021 to $0.0660 / $0.0121 to $0.0877",
              "costShareRange": "17% to 77%",
              "zero": 0,
              "medShare": 0.5443,
              "shareRange": "36% to 96%",
              "pooled": 0.6922,
              "meanReasoning": 0.017969,
              "meanVisible": 0.00799,
              "meanInput": 0.007407,
              "meanTotal": 0.033366,
              "costShare": 0.5385,
              "reasoningPerPass": 0.017969,
              "totalPerPass": 0.033366
            },
            {
              "config": "Claude Fable 5.1 · Claude Code",
              "calls": 24,
              "strict": "24/24 (95% Wilson 86% to 100%)",
              "medOut": 1366,
              "medRea": 889,
              "tokenRange": "318 to 6465 / 75 to 5889",
              "costRange": "$0.0037 to $0.2944 / $0.0322 to $0.3399",
              "costShareRange": "5% to 87%",
              "zero": 0,
              "medShare": 0.6424,
              "shareRange": "23% to 97%",
              "pooled": 0.7417,
              "meanReasoning": 0.053696,
              "meanVisible": 0.018702,
              "meanInput": 0.02091,
              "meanTotal": 0.093308,
              "costShare": 0.5755,
              "reasoningPerPass": 0.053696,
              "totalPerPass": 0.093308
            },
            {
              "config": "GPT-6.1 Sol (medium) · Codex CLI",
              "calls": 16,
              "strict": "16/16 (95% Wilson 81% to 100%)",
              "medOut": 335,
              "medRea": 150,
              "tokenRange": "237 to 1766 / 61 to 839",
              "costRange": "$0.0006 to $0.0084 / $0.0099 to $0.0422",
              "costShareRange": "2% to 20%",
              "zero": 0,
              "medShare": 0.4633,
              "shareRange": "11% to 87%",
              "pooled": 0.429,
              "meanReasoning": 0.002273,
              "meanVisible": 0.003025,
              "meanInput": 0.020339,
              "meanTotal": 0.025637,
              "costShare": 0.0887,
              "reasoningPerPass": 0.002273,
              "totalPerPass": 0.025637
            },
            {
              "config": "GPT-6.1 Sol (high) · Codex CLI",
              "calls": 16,
              "strict": "16/16 (95% Wilson 81% to 100%)",
              "medOut": 436,
              "medRea": 225,
              "tokenRange": "284 to 2569 / 144 to 1750",
              "costRange": "$0.0014 to $0.0175 / $0.0092 to $0.0307",
              "costShareRange": "8% to 60%",
              "zero": 0,
              "medShare": 0.5701,
              "shareRange": "29% to 91%",
              "pooled": 0.5861,
              "meanReasoning": 0.004114,
              "meanVisible": 0.002905,
              "meanInput": 0.008117,
              "meanTotal": 0.015137,
              "costShare": 0.2718,
              "reasoningPerPass": 0.004114,
              "totalPerPass": 0.015137
            }
          ]
        },
        {
          "id": "thinking-bill-effort-cells",
          "title": "Reasoning tokens and list-price cost by effort, every effort-ladder cell (calculation)",
          "columns": [
            {
              "key": "config",
              "label": "Configuration",
              "unit": "text"
            },
            {
              "key": "calls",
              "label": "Calls",
              "unit": "count"
            },
            {
              "key": "strict",
              "label": "Strict passes",
              "unit": "text"
            },
            {
              "key": "medOut",
              "label": "Median output tokens",
              "unit": "tokens"
            },
            {
              "key": "medRea",
              "label": "Median reasoning tokens",
              "unit": "tokens"
            },
            {
              "key": "tokenRange",
              "label": "Output / reasoning tokens, lowest to highest call",
              "unit": "text"
            },
            {
              "key": "costRange",
              "label": "Reasoning / total cost, lowest to highest call (USD, calculation)",
              "unit": "text"
            },
            {
              "key": "costShareRange",
              "label": "Reasoning cost share, lowest to highest call (calculation)",
              "unit": "text"
            },
            {
              "key": "zero",
              "label": "Calls with 0 reasoning tokens",
              "unit": "count"
            },
            {
              "key": "medShare",
              "label": "Median call: reasoning share of output",
              "unit": "rate"
            },
            {
              "key": "shareRange",
              "label": "Share, lowest to highest call",
              "unit": "text"
            },
            {
              "key": "pooled",
              "label": "Pooled share (all reasoning ÷ all output)",
              "unit": "rate"
            },
            {
              "key": "meanReasoning",
              "label": "Reasoning cost per call (USD, calculation)",
              "unit": "usd"
            },
            {
              "key": "meanVisible",
              "label": "Remaining output cost per call (USD, calculation)",
              "unit": "usd"
            },
            {
              "key": "meanInput",
              "label": "Input cost per call (USD)",
              "unit": "usd"
            },
            {
              "key": "meanTotal",
              "label": "Total cost per call (USD)",
              "unit": "usd"
            },
            {
              "key": "costShare",
              "label": "Pooled reasoning share of list-price cost (calculation)",
              "unit": "rate"
            },
            {
              "key": "reasoningPerPass",
              "label": "Reasoning cost per strict pass (USD)",
              "unit": "usd"
            },
            {
              "key": "totalPerPass",
              "label": "Total cost per strict pass (USD)",
              "unit": "usd"
            }
          ],
          "rows": [
            {
              "config": "Claude Sonnet 5.5 (low) · Claude Code",
              "calls": 16,
              "strict": "16/16 (95% Wilson 81% to 100%)",
              "medOut": 667,
              "medRea": 273,
              "tokenRange": "176 to 2263 / 0 to 1489",
              "costRange": "$0.0000 to $0.0149 / $0.0051 to $0.0261",
              "costShareRange": "0% to 71%",
              "zero": 8,
              "medShare": 0.2142,
              "shareRange": "0% to 95%",
              "pooled": 0.5327,
              "meanReasoning": 0.004317,
              "meanVisible": 0.003787,
              "meanInput": 0.004088,
              "meanTotal": 0.012191,
              "costShare": 0.3541,
              "reasoningPerPass": 0.004317,
              "totalPerPass": 0.012191
            },
            {
              "config": "Claude Sonnet 5.5 (medium) · Claude Code",
              "calls": 16,
              "strict": "16/16 (95% Wilson 81% to 100%)",
              "medOut": 770,
              "medRea": 422,
              "tokenRange": "224 to 2836 / 0 to 2093",
              "costRange": "$0.0000 to $0.0209 / $0.0056 to $0.0318",
              "costShareRange": "0% to 75%",
              "zero": 5,
              "medShare": 0.5162,
              "shareRange": "0% to 96%",
              "pooled": 0.6158,
              "meanReasoning": 0.005946,
              "meanVisible": 0.003709,
              "meanInput": 0.003865,
              "meanTotal": 0.01352,
              "costShare": 0.4398,
              "reasoningPerPass": 0.005946,
              "totalPerPass": 0.01352
            },
            {
              "config": "Claude Sonnet 5.5 (high) · Claude Code",
              "calls": 16,
              "strict": "16/16 (95% Wilson 81% to 100%)",
              "medOut": 1192,
              "medRea": 745,
              "tokenRange": "220 to 4187 / 0 to 3610",
              "costRange": "$0.0000 to $0.0361 / $0.0056 to $0.0453",
              "costShareRange": "0% to 80%",
              "zero": 2,
              "medShare": 0.6131,
              "shareRange": "0% to 96%",
              "pooled": 0.7297,
              "meanReasoning": 0.009369,
              "meanVisible": 0.003471,
              "meanInput": 0.003866,
              "meanTotal": 0.016705,
              "costShare": 0.5608,
              "reasoningPerPass": 0.009369,
              "totalPerPass": 0.016705
            },
            {
              "config": "Claude Sonnet 5.5 · Claude Code",
              "calls": 16,
              "strict": "16/16 (95% Wilson 81% to 100%)",
              "medOut": 1054,
              "medRea": 668,
              "tokenRange": "176 to 2185 / 0 to 1614",
              "costRange": "$0.0000 to $0.0161 / $0.0051 to $0.0253",
              "costShareRange": "0% to 74%",
              "zero": 4,
              "medShare": 0.5601,
              "shareRange": "0% to 96%",
              "pooled": 0.6368,
              "meanReasoning": 0.006299,
              "meanVisible": 0.003593,
              "meanInput": 0.004086,
              "meanTotal": 0.013978,
              "costShare": 0.4507,
              "reasoningPerPass": 0.006299,
              "totalPerPass": 0.013978
            },
            {
              "config": "Claude Opus 5.5 (low) · Claude Code",
              "calls": 16,
              "strict": "16/16 (95% Wilson 81% to 100%)",
              "medOut": 594,
              "medRea": 87,
              "tokenRange": "220 to 1569 / 0 to 791",
              "costRange": "$0.0000 to $0.0158 / $0.0108 to $0.0380",
              "costShareRange": "0% to 67%",
              "zero": 7,
              "medShare": 0.3065,
              "shareRange": "0% to 94%",
              "pooled": 0.3785,
              "meanReasoning": 0.005031,
              "meanVisible": 0.008261,
              "meanInput": 0.00786,
              "meanTotal": 0.021152,
              "costShare": 0.2379,
              "reasoningPerPass": 0.005031,
              "totalPerPass": 0.021152
            },
            {
              "config": "Claude Opus 5.5 (medium) · Claude Code",
              "calls": 16,
              "strict": "16/16 (95% Wilson 81% to 100%)",
              "medOut": 853,
              "medRea": 518,
              "tokenRange": "301 to 3075 / 109 to 1993",
              "costRange": "$0.0022 to $0.0399 / $0.0125 to $0.0681",
              "costShareRange": "16% to 74%",
              "zero": 0,
              "medShare": 0.5353,
              "shareRange": "28% to 96%",
              "pooled": 0.609,
              "meanReasoning": 0.01344,
              "meanVisible": 0.00863,
              "meanInput": 0.007405,
              "meanTotal": 0.029475,
              "costShare": 0.456,
              "reasoningPerPass": 0.01344,
              "totalPerPass": 0.029475
            },
            {
              "config": "Claude Opus 5.5 (high) · Claude Code",
              "calls": 16,
              "strict": "16/16 (95% Wilson 81% to 100%)",
              "medOut": 1052,
              "medRea": 614,
              "tokenRange": "285 to 4052 / 103 to 3301",
              "costRange": "$0.0021 to $0.0660 / $0.0121 to $0.0877",
              "costShareRange": "17% to 77%",
              "zero": 0,
              "medShare": 0.5369,
              "shareRange": "36% to 96%",
              "pooled": 0.6865,
              "meanReasoning": 0.018034,
              "meanVisible": 0.008236,
              "meanInput": 0.007407,
              "meanTotal": 0.033677,
              "costShare": 0.5355,
              "reasoningPerPass": 0.018034,
              "totalPerPass": 0.033677
            },
            {
              "config": "Claude Opus 5.5 · Claude Code",
              "calls": 16,
              "strict": "16/16 (95% Wilson 81% to 100%)",
              "medOut": 945,
              "medRea": 538,
              "tokenRange": "323 to 2531 / 100 to 1778",
              "costRange": "$0.0020 to $0.0356 / $0.0131 to $0.0572",
              "costShareRange": "10% to 72%",
              "zero": 0,
              "medShare": 0.5435,
              "shareRange": "31% to 95%",
              "pooled": 0.6221,
              "meanReasoning": 0.013104,
              "meanVisible": 0.007961,
              "meanInput": 0.00786,
              "meanTotal": 0.028925,
              "costShare": 0.453,
              "reasoningPerPass": 0.013104,
              "totalPerPass": 0.028925
            },
            {
              "config": "GPT-6.1 Sol (low) · Codex CLI",
              "calls": 16,
              "strict": "16/16 (95% Wilson 81% to 100%)",
              "medOut": 284,
              "medRea": 63,
              "tokenRange": "172 to 1310 / 0 to 458",
              "costRange": "$0.0000 to $0.0046 / $0.0092 to $0.0264",
              "costShareRange": "0% to 22%",
              "zero": 4,
              "medShare": 0.2351,
              "shareRange": "0% to 84%",
              "pooled": 0.2909,
              "meanReasoning": 0.001223,
              "meanVisible": 0.00298,
              "meanInput": 0.008635,
              "meanTotal": 0.012837,
              "costShare": 0.0952,
              "reasoningPerPass": 0.001223,
              "totalPerPass": 0.012837
            },
            {
              "config": "GPT-6.1 Sol (medium) · Codex CLI",
              "calls": 16,
              "strict": "16/16 (95% Wilson 81% to 100%)",
              "medOut": 335,
              "medRea": 150,
              "tokenRange": "237 to 1766 / 61 to 839",
              "costRange": "$0.0006 to $0.0084 / $0.0099 to $0.0422",
              "costShareRange": "2% to 20%",
              "zero": 0,
              "medShare": 0.4633,
              "shareRange": "11% to 87%",
              "pooled": 0.429,
              "meanReasoning": 0.002273,
              "meanVisible": 0.003025,
              "meanInput": 0.020339,
              "meanTotal": 0.025637,
              "costShare": 0.0887,
              "reasoningPerPass": 0.002273,
              "totalPerPass": 0.025637
            },
            {
              "config": "GPT-6.1 Sol (high) · Codex CLI",
              "calls": 16,
              "strict": "16/16 (95% Wilson 81% to 100%)",
              "medOut": 436,
              "medRea": 225,
              "tokenRange": "284 to 2569 / 144 to 1750",
              "costRange": "$0.0014 to $0.0175 / $0.0092 to $0.0307",
              "costShareRange": "8% to 60%",
              "zero": 0,
              "medShare": 0.5701,
              "shareRange": "29% to 91%",
              "pooled": 0.5861,
              "meanReasoning": 0.004114,
              "meanVisible": 0.002905,
              "meanInput": 0.008117,
              "meanTotal": 0.015137,
              "costShare": 0.2718,
              "reasoningPerPass": 0.004114,
              "totalPerPass": 0.015137
            }
          ]
        },
        {
          "id": "thinking-bill-short-cells",
          "title": "Reasoning tokens and list-price cost per call, five short tasks (calculation)",
          "columns": [
            {
              "key": "config",
              "label": "Configuration",
              "unit": "text"
            },
            {
              "key": "calls",
              "label": "Calls",
              "unit": "count"
            },
            {
              "key": "strict",
              "label": "Strict passes",
              "unit": "text"
            },
            {
              "key": "medOut",
              "label": "Median output tokens",
              "unit": "tokens"
            },
            {
              "key": "medRea",
              "label": "Median reasoning tokens",
              "unit": "tokens"
            },
            {
              "key": "tokenRange",
              "label": "Output / reasoning tokens, lowest to highest call",
              "unit": "text"
            },
            {
              "key": "costRange",
              "label": "Reasoning / total cost, lowest to highest call (USD, calculation)",
              "unit": "text"
            },
            {
              "key": "costShareRange",
              "label": "Reasoning cost share, lowest to highest call (calculation)",
              "unit": "text"
            },
            {
              "key": "zero",
              "label": "Calls with 0 reasoning tokens",
              "unit": "count"
            },
            {
              "key": "medShare",
              "label": "Median call: reasoning share of output",
              "unit": "rate"
            },
            {
              "key": "shareRange",
              "label": "Share, lowest to highest call",
              "unit": "text"
            },
            {
              "key": "pooled",
              "label": "Pooled share (all reasoning ÷ all output)",
              "unit": "rate"
            },
            {
              "key": "meanReasoning",
              "label": "Reasoning cost per call (USD, calculation)",
              "unit": "usd"
            },
            {
              "key": "meanVisible",
              "label": "Remaining output cost per call (USD, calculation)",
              "unit": "usd"
            },
            {
              "key": "meanInput",
              "label": "Input cost per call (USD)",
              "unit": "usd"
            },
            {
              "key": "meanTotal",
              "label": "Total cost per call (USD)",
              "unit": "usd"
            },
            {
              "key": "costShare",
              "label": "Pooled reasoning share of list-price cost (calculation)",
              "unit": "rate"
            },
            {
              "key": "reasoningPerPass",
              "label": "Reasoning cost per strict pass (USD)",
              "unit": "usd"
            },
            {
              "key": "totalPerPass",
              "label": "Total cost per strict pass (USD)",
              "unit": "usd"
            }
          ],
          "rows": [
            {
              "config": "Claude Haiku 4.5 · Claude Code",
              "calls": 15,
              "strict": "15/15 (95% Wilson 80% to 100%)",
              "medOut": 367,
              "medRea": 297,
              "tokenRange": "275 to 2851 / 212 to 2643",
              "costRange": "$0.0011 to $0.0132 / $0.0051 to $0.0180",
              "costShareRange": "20% to 74%",
              "zero": 0,
              "medShare": 0.9019,
              "shareRange": "73% to 98%",
              "pooled": 0.9049,
              "meanReasoning": 0.004133,
              "meanVisible": 0.000434,
              "meanInput": 0.00379,
              "meanTotal": 0.008358,
              "costShare": 0.4945,
              "reasoningPerPass": 0.004133,
              "totalPerPass": 0.008358
            },
            {
              "config": "Claude Sonnet 5.5 · Claude Code",
              "calls": 15,
              "strict": "12/15 (95% Wilson 55% to 93%)",
              "medOut": 107,
              "medRea": 0,
              "tokenRange": "44 to 745 / 0 to 542",
              "costRange": "$0.0000 to $0.0054 / $0.0034 to $0.0102",
              "costShareRange": "0% to 53%",
              "zero": 12,
              "medShare": 0,
              "shareRange": "0% to 73%",
              "pooled": 0.4468,
              "meanReasoning": 0.000882,
              "meanVisible": 0.001092,
              "meanInput": 0.003015,
              "meanTotal": 0.004989,
              "costShare": 0.1768,
              "reasoningPerPass": 0.001103,
              "totalPerPass": 0.006236
            },
            {
              "config": "Claude Opus 5.5 · Claude Code",
              "calls": 15,
              "strict": "15/15 (95% Wilson 80% to 100%)",
              "medOut": 64,
              "medRea": 0,
              "tokenRange": "44 to 853 / 0 to 616",
              "costRange": "$0.0000 to $0.0123 / $0.0059 to $0.0223",
              "costShareRange": "0% to 55%",
              "zero": 9,
              "medShare": 0,
              "shareRange": "0% to 93%",
              "pooled": 0.5912,
              "meanReasoning": 0.002585,
              "meanVisible": 0.001788,
              "meanInput": 0.005714,
              "meanTotal": 0.010088,
              "costShare": 0.2563,
              "reasoningPerPass": 0.002585,
              "totalPerPass": 0.010088
            },
            {
              "config": "Claude Opus 5.5 (low) · Claude Code",
              "calls": 15,
              "strict": "15/15 (95% Wilson 80% to 100%)",
              "medOut": 64,
              "medRea": 0,
              "tokenRange": "44 to 637 / 0 to 405",
              "costRange": "$0.0000 to $0.0081 / $0.0058 to $0.0179",
              "costShareRange": "0% to 45%",
              "zero": 9,
              "medShare": 0,
              "shareRange": "0% to 93%",
              "pooled": 0.4098,
              "meanReasoning": 0.001253,
              "meanVisible": 0.001805,
              "meanInput": 0.005232,
              "meanTotal": 0.00829,
              "costShare": 0.1512,
              "reasoningPerPass": 0.001253,
              "totalPerPass": 0.00829
            },
            {
              "config": "Claude Opus 5.5 (high) · Claude Code",
              "calls": 15,
              "strict": "15/15 (95% Wilson 80% to 100%)",
              "medOut": 78,
              "medRea": 34,
              "tokenRange": "44 to 1094 / 0 to 857",
              "costRange": "$0.0000 to $0.0171 / $0.0059 to $0.0271",
              "costShareRange": "0% to 63%",
              "zero": 7,
              "medShare": 0.4359,
              "shareRange": "0% to 93%",
              "pooled": 0.6562,
              "meanReasoning": 0.003448,
              "meanVisible": 0.001807,
              "meanInput": 0.005233,
              "meanTotal": 0.010488,
              "costShare": 0.3288,
              "reasoningPerPass": 0.003448,
              "totalPerPass": 0.010488
            },
            {
              "config": "Claude Fable 5.1 · Claude Code",
              "calls": 15,
              "strict": "15/15 (95% Wilson 80% to 100%)",
              "medOut": 64,
              "medRea": 0,
              "tokenRange": "4 to 908 / 0 to 672",
              "costRange": "$0.0000 to $0.0336 / $0.0049 to $0.0584",
              "costShareRange": "0% to 67%",
              "zero": 12,
              "medShare": 0,
              "shareRange": "0% to 74%",
              "pooled": 0.5724,
              "meanReasoning": 0.005957,
              "meanVisible": 0.00445,
              "meanInput": 0.010133,
              "meanTotal": 0.020539,
              "costShare": 0.29,
              "reasoningPerPass": 0.005957,
              "totalPerPass": 0.020539
            },
            {
              "config": "GPT-6.1 Sol (low) · Codex CLI",
              "calls": 10,
              "strict": "10/10 (95% Wilson 72% to 100%)",
              "medOut": 42,
              "medRea": 20,
              "tokenRange": "28 to 225 / 0 to 103",
              "costRange": "$0.0000 to $0.0010 / $0.0074 to $0.0265",
              "costShareRange": "0% to 11%",
              "zero": 4,
              "medShare": 0.2528,
              "shareRange": "0% to 71%",
              "pooled": 0.3132,
              "meanReasoning": 0.000332,
              "meanVisible": 0.000728,
              "meanInput": 0.008924,
              "meanTotal": 0.009984,
              "costShare": 0.0333,
              "reasoningPerPass": 0.000332,
              "totalPerPass": 0.009984
            },
            {
              "config": "GPT-6.1 Sol (medium) · Codex CLI",
              "calls": 15,
              "strict": "15/15 (95% Wilson 80% to 100%)",
              "medOut": 42,
              "medRea": 20,
              "tokenRange": "27 to 295 / 0 to 137",
              "costRange": "$0.0000 to $0.0014 / $0.0054 to $0.0269",
              "costShareRange": "0% to 23%",
              "zero": 6,
              "medShare": 0.4105,
              "shareRange": "0% to 71%",
              "pooled": 0.4142,
              "meanReasoning": 0.000512,
              "meanVisible": 0.000724,
              "meanInput": 0.014404,
              "meanTotal": 0.01564,
              "costShare": 0.0327,
              "reasoningPerPass": 0.000512,
              "totalPerPass": 0.01564
            },
            {
              "config": "GPT-6.1 Sol (high) · Codex CLI",
              "calls": 15,
              "strict": "15/15 (95% Wilson 80% to 100%)",
              "medOut": 42,
              "medRea": 21,
              "tokenRange": "28 to 457 / 0 to 289",
              "costRange": "$0.0000 to $0.0029 / $0.0066 to $0.0281",
              "costShareRange": "0% to 29%",
              "zero": 6,
              "medShare": 0.5806,
              "shareRange": "0% to 76%",
              "pooled": 0.5814,
              "meanReasoning": 0.001007,
              "meanVisible": 0.000725,
              "meanInput": 0.011484,
              "meanTotal": 0.013216,
              "costShare": 0.0762,
              "reasoningPerPass": 0.001007,
              "totalPerPass": 0.013216
            }
          ]
        },
        {
          "id": "thinking-bill-time-link",
          "title": "Does reasoning track time? Rank correlation and slope per model (calculation)",
          "columns": [
            {
              "key": "model",
              "label": "Model and route",
              "unit": "text"
            },
            {
              "key": "calls",
              "label": "Calls",
              "unit": "count"
            },
            {
              "key": "rho",
              "label": "Spearman: reasoning tokens vs time",
              "unit": "score"
            },
            {
              "key": "rhoCi",
              "label": "Leave-one-task-out range (not a 95% interval)",
              "unit": "text"
            },
            {
              "key": "rhoOut",
              "label": "Spearman: output tokens vs time",
              "unit": "score"
            },
            {
              "key": "perK",
              "label": "Seconds per 1,000 reasoning tokens (all calls)",
              "unit": "seconds"
            },
            {
              "key": "withinK",
              "label": "Seconds per 1,000 reasoning tokens (within task)",
              "unit": "seconds"
            },
            {
              "key": "withinCi",
              "label": "Within-task slope, leave-one-task-out range",
              "unit": "text"
            },
            {
              "key": "medRea",
              "label": "Median reasoning tokens",
              "unit": "tokens"
            },
            {
              "key": "medTime",
              "label": "Median total time (s)",
              "unit": "seconds"
            },
            {
              "key": "timeRange",
              "label": "Total time, lowest to highest call (s; not an interval)",
              "unit": "text"
            }
          ],
          "rows": [
            {
              "model": "Claude Haiku 4.5 · Claude Code",
              "calls": 24,
              "rho": 0.982,
              "rhoCi": "0.98 to 0.99",
              "rhoOut": 0.986,
              "perK": 8.408,
              "withinK": 8.817,
              "withinCi": "8.6 to 9.1",
              "medRea": 4556,
              "medTime": 39.01,
              "timeRange": "15.27 to 75.13"
            },
            {
              "model": "Claude Sonnet 5.5 · Claude Code",
              "calls": 72,
              "rho": 0.921,
              "rhoCi": "0.88 to 0.94",
              "rhoOut": 0.94,
              "perK": 9.28,
              "withinK": 7.649,
              "withinCi": "7.4 to 7.8",
              "medRea": 586,
              "medTime": 7.75,
              "timeRange": "2.26 to 35.81"
            },
            {
              "model": "Claude Opus 5.5 · Claude Code",
              "calls": 80,
              "rho": 0.852,
              "rhoCi": "0.82 to 0.89",
              "rhoOut": 0.97,
              "perK": 13.091,
              "withinK": 11.754,
              "withinCi": "5.7 to 12.7",
              "medRea": 511,
              "medTime": 9.01,
              "timeRange": "3.34 to 63.00"
            },
            {
              "model": "Claude Fable 5.1 · Claude Code",
              "calls": 24,
              "rho": 0.924,
              "rhoCi": "0.89 to 0.93",
              "rhoOut": 0.941,
              "perK": 14.39,
              "withinK": 14.454,
              "withinCi": "14.4 to 22.6",
              "medRea": 889,
              "medTime": 16.13,
              "timeRange": "4.46 to 90.00"
            },
            {
              "model": "GPT-6.1 Sol · Codex CLI",
              "calls": 48,
              "rho": 0.556,
              "rhoCi": "0.34 to 0.65",
              "rhoOut": 0.85,
              "perK": 50.374,
              "withinK": 36.351,
              "withinCi": "31.8 to 36.7",
              "medRea": 189,
              "medTime": 14.51,
              "timeRange": "7.94 to 92.20"
            }
          ]
        },
        {
          "id": "thinking-bill-sum-check",
          "title": "Consistency check for reasoning within output tokens (calculation)",
          "columns": [
            {
              "key": "route",
              "label": "CLI route",
              "unit": "text"
            },
            {
              "key": "calls",
              "label": "Calls",
              "unit": "count"
            },
            {
              "key": "withReasoning",
              "label": "Calls with more than 0 reasoning tokens",
              "unit": "count"
            },
            {
              "key": "zero",
              "label": "Calls that reported 0 (kept as 0)",
              "unit": "count"
            },
            {
              "key": "over",
              "label": "Calls with reasoning above output",
              "unit": "count"
            },
            {
              "key": "slope",
              "label": "Output tokens gained per extra reasoning token (within task)",
              "unit": "ratio"
            },
            {
              "key": "slopeCi",
              "label": "Leave-one-task-out range (not a 95% interval)",
              "unit": "text"
            },
            {
              "key": "cpvisible",
              "label": "Characters per token of output minus reasoning",
              "unit": "score"
            },
            {
              "key": "cpoutZero",
              "label": "Characters per output token, calls with 0 reasoning",
              "unit": "score"
            },
            {
              "key": "cpoutWith",
              "label": "Characters per output token, calls with reasoning",
              "unit": "score"
            },
            {
              "key": "verdict",
              "label": "Reasoning is part of output",
              "unit": "text"
            }
          ],
          "rows": [
            {
              "route": "Claude Code",
              "calls": 290,
              "withReasoning": 212,
              "zero": 78,
              "over": 0,
              "slope": 0.993,
              "slopeCi": "0.982 to 1.009",
              "cpvisible": 1.94,
              "cpoutZero": 1.98,
              "cpoutWith": 0.45,
              "verdict": "consistent with inclusion; not proof"
            },
            {
              "route": "Codex CLI",
              "calls": 88,
              "withReasoning": 68,
              "zero": 20,
              "over": 0,
              "slope": 0.967,
              "slopeCi": "0.965 to 0.986",
              "cpvisible": 2.99,
              "cpoutZero": 2.76,
              "cpoutWith": 1.36,
              "verdict": "consistent with inclusion; not proof"
            }
          ]
        }
      ],
      "related": [
        "effort-ladder",
        "hard-model-head-to-head"
      ]
    },
    {
      "slug": "llm-speed-anatomy",
      "title": "Where the seconds go: first text, output speed and prompt size for 6 LLMs",
      "seoTitle": "LLM speed: first text and tokens per second through CLIs",
      "description": "60 calls on 6 models: time to first text, tokens per second, and what a 1k, 16k or 64k prompt adds. Claude Code and Codex CLI.",
      "question": "For 6 models run through their own coding CLIs (Haiku, Sonnet, Opus, Fable, Sol (low) and Luna (low)), how long until the first text, how fast does text stream after it, and what does a longer prompt add?",
      "answer": "We measured CLI start-up, first text and total time on one 250-line answer and a ledger lookup at three prompt sizes. There were 60 counted calls: 4 per model in part A and 3 per size in part B. The receipts do not separate provider queueing, prompt processing and reasoning time. Time to first text, median (range; n calls): Sonnet 2.0 s (0.9 to 4.1; n = 4). Opus 2.0 s (1.7 to 2.4; n = 4). Luna (low) 3.3 s (3.2 to 3.5; n = 4). Sol (low) 3.5 s (2.7 to 4.4; n = 4). Haiku 4.0 s (2.8 to 6.4; n = 4). Fable 4.4 s (2.3 to 4.6; n = 4). Opus’s slowest call (2.4 s) was faster than the fastest call of Luna (low) (3.2 s), Sol (low) (2.7 s) and Haiku (2.8 s). Sonnet and Fable overlap it. There are 4 calls per model. The samples are below the protocol’s five-call minimum for naming a speed winner. Characters per second is a calculation: correct reply length ÷ time from first text to call end. Median (range; n exact replies): Haiku 547 (546 to 548; n = 3). Luna (low) 524 (225 to 1,052; n = 4). Sonnet 517 (513 to 519; n = 4). Opus 347 (345 to 349; n = 4). Sol (low) 323 (291 to 327; n = 4). Fable 273 (270 to 293; n = 4). Haiku and Fable have non-overlapping observed speed ranges (3 and 4 exact replies). Both samples are below the protocol’s five-call minimum for naming a speed winner. Luna (low)’s 4 exact replies ran from 225 to 1,052 characters per second, so its median hides this large spread. These few calls cannot establish two speed groups. Token rates use different token units across models. The same exact reply has 4,327 characters. Median visible tokens: 1,066 for Sol (low) (n = 4) and Luna (low) (n = 4). 1,210 for Haiku (n = 3). 1,941 for Sonnet (n = 4), Opus (n = 4) and Fable (n = 4). Visible tokens per second (calculation), median (range; n calls): Sonnet 232 (230 to 233; n = 4). Opus 156 (155 to 156; n = 4). Haiku 153 (153 to 216; n = 4). Luna (low) 129 (55 to 259; n = 4). Fable 123 (121 to 131; n = 4). Sol (low) 80 (72 to 80; n = 4). These ranges show individual calls, not confidence intervals. Of the 24 part A replies, 23 matched all 250 lines exactly; 1 was wrong (Haiku, 288 lines), and every call stays in the timings. First-text medians differ by prompt size (calculation: difference of medians, 64k minus 1k): Haiku +0.9 s (1.9 s to 2.8 s). Sonnet +1.6 s (1.4 s to 3.1 s, ranges overlap). Opus +0.3 s (1.5 s to 1.8 s, ranges overlap). Sol (low) +0.6 s (3.4 s to 3.9 s, ranges overlap). Exact lookup answers (95% Wilson intervals): Haiku 9/9 (70% to 100%). Sonnet 9/9 (70% to 100%). Opus 5/9 (27% to 81%). Sol (low) 9/9 (70% to 100%). All 4 models’ intervals overlap; this sample cannot rank their lookup rates. 4 Opus replies echoed the ledger line and then gave the right number last. The declared scoring counts these as wrong answers. The table shows the post-hoc reading. Cache reads did not grow with the ledger for any model. This fits no ledger reuse, but the counts do not prove which text the provider cached.",
      "date": "2026-10-07",
      "updated": "2026-10-07",
      "tags": [
        "latency",
        "time-to-first-token",
        "tokens-per-second",
        "output-speed",
        "prompt-size",
        "claude-code",
        "codex-cli",
        "claude-haiku",
        "claude-sonnet",
        "claude-opus",
        "claude-fable",
        "gpt-6-1-sol",
        "gpt-6-luna"
      ],
      "method": [
        "The protocol file birth precedes the first probe and counted call. Later edits have no frozen versions. Two probes checked the routes and helped calibrate ledger sizes. They stay outside all cells and charts.",
        "Part A used 24 calls: 4 per model. The prompt asked for 250 numbers in words, one per line. The check compares the reply with all 250 expected lines. Models: Haiku, Sonnet, Opus, Fable, Sol (low) and Luna (low).\n\nControls ran before the first counted call. The reference passes. All 8/8 wrapped references are format misses. All 11/11 planted wrong answers fail.",
        "Part B used 36 calls: 3 per size for each model. Models: Haiku, Sonnet, Opus and Sol (low). Each call asks one exact lookup question about a synthetic stock ledger. Ledger sizes: 18 lines for 1k, 327 lines for 16k and 1,315 lines for 64k.\n\nThe 1k, 16k and 64k labels are approximate targets from probe calibration calculations, using estimated Haiku prefix counts. A new seed changes each ledger. Some fixed text repeats. Cache counts do not identify cached text.",
        "First text means the first non-empty text after CLI start. The time includes CLI start-up and work before the first word. The table also subtracts the CLI ready time (calculation).",
        "Output speed is a calculation. Visible tokens equal output tokens minus reported reasoning tokens. When the CLI reports no reasoning count, the calculation uses zero. That does not prove the model did no reasoning. Divide visible tokens by the time from first text to call end. \n\nThe plain form divides all output tokens by that time. It can overstate visible speed when reasoning comes first.",
        "The harness uses the hard head-to-head machinery. It disables tools, MCP servers and saved sessions. Claude Code uses default effort; Codex CLI uses low effort. Calls run one at a time on one Mac.\n\nThe timeout was 300 s. Claude Code had a 16,000-token output cap. Codex had no output-token cap. CLI versions: 2.1.286 (Claude Code) and codex-cli 0.160.0.",
        "Stop rules require a stop at the first usage-limit text. The gate checks other study markers before each batch. No stored batch stopped or trimmed a cell. No counted call duplicates another. \n\nA sequence process ended at the gate. Codex part B ran later. Every counted call completed.",
        "Cache-read share is a calculation: cache-read tokens divided by reported input tokens for each call. The cache table shows the median, observed range and n for each cell."
      ],
      "caveats": [
        "4 calls per model in part A and 3 per size in part B. Medians of so few calls move with one slow call, and the ranges are not confidence intervals. The 95% Wilson intervals on the lookup rates are wide.",
        "First text includes CLI start-up and reasoning time. It is not the API’s time to first token; a direct API call would skip the CLI start-up (the CLI reported ready after a median 0.33 s to 0.84 s per cell).",
        "Claude Code ran models at their default effort; Codex CLI ran at low effort. Effort levels are not one scale, and a CLI adds its own system prompt, so a row mixes a model and its CLI.",
        "Output speed is a calculation from reported token counts, the length of the reply and the clock. It depends on how each CLI counts reasoning, and part A has one prompt: other text, other days or an API route can differ.",
        "All 36 counted ledger prompts differ. Their input counts do not compare tokenizers on the same text. The size calibration uses prefix estimates from earlier short prompts. Those calculations do not measure the prefix in each counted call.",
        "The tasks hit a ceiling. Part A passed 23/24. In part B, 3 of 4 models passed every lookup. See the rate stats and chart for 95% intervals. This small task set cannot rank general capability.",
        "One shared Mac: the gate checked other study markers before each batch. No full host-load record proves that all other work stopped.",
        "Prompt sizes were calibrated after two probes. The cases are synthetic and tuned to this measurement, not a sample of real workloads.",
        "Amendment 3 says 06:05 UTC and claims to precede Codex part B. Those calls ran at 06:03:43 to 06:04:19 UTC. The same-file amendment is retrospective; its date does not prove advance declaration.",
        "The controls receipt was overwritten. The surviving checks ran after the probes and before counted calls; an earlier control run cannot be audited.",
        "Each size used different ledger text. Prompt-size differences also include content and provider-load changes; they do not isolate a cause.",
        "The failed Haiku reply contained tool-shaped text and extra output. No structured tool event was recorded; it stays a strict failure and remains in timings.",
        "One Mac, one network, two sessions on one night. Provider load changes timings from hour to hour. The Codex CLI part B batch ran about 4 hours after the batches before it (a run script stopped while it waited for another run to finish timing), so its timings come from a later hour."
      ],
      "sourceIds": [
        "agent-speed-anatomy"
      ],
      "hero": {
        "statIds": [
          "speed-anatomy-token-count-ratio"
        ]
      },
      "stats": [
        {
          "id": "speed-anatomy-calls",
          "label": "Counted calls in the speed anatomy study",
          "value": 60,
          "unit": "calls",
          "display": "60 calls (24 in part A, 36 in part B; 43 through Claude Code, 17 through Codex CLI)",
          "n": 60
        },
        {
          "id": "speed-anatomy-part-a-exact",
          "label": "Part A replies that matched all 250 lines exactly (strict)",
          "value": 0.9583,
          "unit": "rate",
          "display": "96% (23/24)",
          "n": 24,
          "ci": [
            0.7976,
            0.9926
          ],
          "note": "0 more were right but wrapped or laid out differently (format misses); 1 was wrong."
        },
        {
          "id": "speed-anatomy-lookup-exact",
          "label": "Part B lookups answered exactly (strict)",
          "value": 0.8889,
          "unit": "rate",
          "display": "89% (32/36)",
          "n": 36,
          "ci": [
            0.7469,
            0.9559
          ],
          "note": "0 more were right with extra words (format misses); 4 were wrong."
        },
        {
          "id": "speed-anatomy-token-count-ratio",
          "label": "The same 4,327-character reply: most tokens ÷ fewest tokens across models (calculation)",
          "value": 1.8,
          "unit": "ratio",
          "display": "1.8x",
          "n": 23,
          "note": "Median visible tokens of exact replies per model (observed token range; n): Haiku 1,210 (1,210 to 1,210; n = 3). Sonnet 1,941 (1,941 to 1,941; n = 4). Opus 1,941 (1,941 to 1,941; n = 4). Fable 1,941 (1,941 to 1,941; n = 4). Sol (low) 1,066 (1,066 to 1,066; n = 4). Luna (low) 1,066 (1,066 to 1,066; n = 4). Each vendor counts the same text differently, so tokens per second do not compare across vendors. A calculation, not a run."
        },
        {
          "id": "speed-anatomy-lookup-last-line",
          "label": "Part B replies whose last line was exactly the right number (post-hoc reading, not a pass)",
          "value": 1,
          "unit": "rate",
          "display": "100% (36/36)",
          "n": 36,
          "ci": [
            0.9036,
            1
          ],
          "note": "A reading declared after the Claude part B batch (protocol Amendment 2), derived from the stored replies. The strict score above stays the headline; this one is never counted as a pass or a format miss."
        },
        {
          "id": "speed-anatomy-cache-reads-grew",
          "label": "Models whose cache reads grew with the ledger size",
          "value": 0,
          "unit": "count",
          "display": "0 of 4",
          "n": 36,
          "note": "A new ledger for every call. The descriptive threshold is the 1k-prompt maximum plus 10% and 128 tokens. Growth does not prove ledger reuse; the cache table lists every cell."
        },
        {
          "id": "speed-anatomy-size-cost-haiku",
          "label": "Haiku: extra time to first text at 64k vs 1k (calculation)",
          "value": 0.86,
          "unit": "seconds",
          "display": "+0.9 s",
          "n": 6,
          "note": "Difference of two medians (2.78 s minus 1.93 s); the ranges are 1.85 to 2.04 s and 2.45 to 2.89 s. A calculation, not a run. Each call used separately seeded text. The table shows reported input counts, not a matched-text tokenizer comparison."
        },
        {
          "id": "speed-anatomy-size-cost-sonnet",
          "label": "Sonnet: extra time to first text at 64k vs 1k (calculation)",
          "value": 1.62,
          "unit": "seconds",
          "display": "+1.6 s",
          "n": 6,
          "note": "Difference of two medians (3.07 s minus 1.45 s); the ranges are 1.23 to 1.72 s and 1.38 to 3.61 s and overlap. A calculation, not a run. Each call used separately seeded text. The table shows reported input counts, not a matched-text tokenizer comparison."
        },
        {
          "id": "speed-anatomy-size-cost-opus",
          "label": "Opus: extra time to first text at 64k vs 1k (calculation)",
          "value": 0.28,
          "unit": "seconds",
          "display": "+0.3 s",
          "n": 6,
          "note": "Difference of two medians (1.79 s minus 1.51 s); the ranges are 1.46 to 2.01 s and 1.72 to 3.72 s and overlap. A calculation, not a run. Each call used separately seeded text. The table shows reported input counts, not a matched-text tokenizer comparison."
        },
        {
          "id": "speed-anatomy-size-cost-sol-low",
          "label": "Sol (low): extra time to first text at 64k vs 1k (calculation)",
          "value": 0.57,
          "unit": "seconds",
          "display": "+0.6 s",
          "n": 6,
          "note": "Difference of two medians (3.93 s minus 3.36 s); the ranges are 3.36 to 4.75 s and 3.42 to 4.38 s and overlap. A calculation, not a run. Each call used separately seeded text. The table shows reported input counts, not a matched-text tokenizer comparison."
        }
      ],
      "charts": [
        {
          "id": "speed-anatomy-first-text",
          "title": "Time to first text: a 250-line answer, six models",
          "subtitle": "Median of 4 calls per model; whiskers = fastest and slowest call",
          "kind": "dot-range",
          "unit": "seconds",
          "yLabel": "Seconds to first text",
          "series": [
            {
              "name": "Time to first text",
              "points": [
                {
                  "label": "Claude Haiku 4.5 · Claude Code",
                  "value": 4,
                  "lo": 2.84,
                  "hi": 6.38,
                  "n": 4
                },
                {
                  "label": "Claude Sonnet 5.5 · Claude Code",
                  "value": 1.96,
                  "lo": 0.88,
                  "hi": 4.09,
                  "n": 4
                },
                {
                  "label": "Claude Opus 5.5 · Claude Code",
                  "value": 1.97,
                  "lo": 1.7,
                  "hi": 2.35,
                  "n": 4
                },
                {
                  "label": "Claude Fable 5.1 · Claude Code",
                  "value": 4.43,
                  "lo": 2.27,
                  "hi": 4.64,
                  "n": 4
                },
                {
                  "label": "GPT-6.1 Sol (low) · Codex CLI",
                  "value": 3.52,
                  "lo": 2.75,
                  "hi": 4.42,
                  "n": 4
                },
                {
                  "label": "GPT-6 Luna (low) · Codex CLI",
                  "value": 3.3,
                  "lo": 3.19,
                  "hi": 3.47,
                  "n": 4
                }
              ]
            }
          ],
          "note": "Dot = median; whiskers = fastest and slowest call (a range, not a confidence interval). The clock starts when the CLI starts, so first text includes CLI start-up (see chart cli-startup-tax) and any reasoning before the first word. One Mac, one network, two sessions on one night.",
          "whisker": "minmax",
          "sourceIds": [
            "agent-speed-anatomy"
          ]
        },
        {
          "id": "speed-anatomy-output-speed",
          "title": "Output speed after the first text: visible tokens per second (calculation)",
          "subtitle": "Median of 4 calls per model; whiskers = slowest and fastest call",
          "kind": "dot-range",
          "unit": "tokens",
          "yLabel": "Visible tokens per second",
          "series": [
            {
              "name": "Visible tokens per second",
              "points": [
                {
                  "label": "Claude Haiku 4.5 · Claude Code",
                  "value": 153.2,
                  "lo": 152.6,
                  "hi": 216.1,
                  "n": 4
                },
                {
                  "label": "Claude Sonnet 5.5 · Claude Code",
                  "value": 231.7,
                  "lo": 230.3,
                  "hi": 233,
                  "n": 4
                },
                {
                  "label": "Claude Opus 5.5 · Claude Code",
                  "value": 155.5,
                  "lo": 154.6,
                  "hi": 156.4,
                  "n": 4
                },
                {
                  "label": "Claude Fable 5.1 · Claude Code",
                  "value": 122.6,
                  "lo": 120.9,
                  "hi": 131.4,
                  "n": 4
                },
                {
                  "label": "GPT-6.1 Sol (low) · Codex CLI",
                  "value": 79.6,
                  "lo": 71.6,
                  "hi": 80.5,
                  "n": 4
                },
                {
                  "label": "GPT-6 Luna (low) · Codex CLI",
                  "value": 129.1,
                  "lo": 55.5,
                  "hi": 259.1,
                  "n": 4
                }
              ]
            }
          ],
          "note": "Calculation, not a measurement: visible output tokens (the CLI’s output token count minus its reported reasoning tokens) divided by the time from the first text to the end of the call. The denominator includes the CLI’s exit overhead; its size is not measured here. Whiskers are a range of calls, not a confidence interval. The calculation removes reported reasoning tokens. These receipts do not locate all reasoning in time.",
          "whisker": "minmax",
          "sourceIds": [
            "agent-speed-anatomy"
          ]
        },
        {
          "id": "speed-anatomy-chars-per-second",
          "title": "Output speed in characters per second after the first text (calculation)",
          "subtitle": "Only replies that matched all 250 lines; median per model, whiskers = slowest and fastest call",
          "kind": "dot-range",
          "unit": "count",
          "yLabel": "Characters per second",
          "series": [
            {
              "name": "Characters per second",
              "points": [
                {
                  "label": "Claude Haiku 4.5 · Claude Code",
                  "value": 547,
                  "lo": 546,
                  "hi": 548,
                  "n": 3
                },
                {
                  "label": "Claude Sonnet 5.5 · Claude Code",
                  "value": 517,
                  "lo": 513,
                  "hi": 519,
                  "n": 4
                },
                {
                  "label": "Claude Opus 5.5 · Claude Code",
                  "value": 347,
                  "lo": 345,
                  "hi": 349,
                  "n": 4
                },
                {
                  "label": "Claude Fable 5.1 · Claude Code",
                  "value": 273,
                  "lo": 270,
                  "hi": 293,
                  "n": 4
                },
                {
                  "label": "GPT-6.1 Sol (low) · Codex CLI",
                  "value": 323,
                  "lo": 291,
                  "hi": 327,
                  "n": 4
                },
                {
                  "label": "GPT-6 Luna (low) · Codex CLI",
                  "value": 524,
                  "lo": 225,
                  "hi": 1052,
                  "n": 4
                }
              ]
            }
          ],
          "note": "Calculation, not a measurement: the characters of a correct reply (the same text for every model) divided by the time from the first text to the end of the call. Each vendor counts the same text as a different number of tokens, so characters compare across models where tokens do not. The CLI’s exit time is inside that time. Whiskers are a range of calls, not a confidence interval.",
          "whisker": "minmax",
          "sourceIds": [
            "agent-speed-anatomy"
          ]
        },
        {
          "id": "speed-anatomy-prompt-size",
          "title": "Time to first text as the prompt grows",
          "subtitle": "Median of 3 calls per size; whiskers = fastest and slowest call",
          "kind": "line",
          "unit": "seconds",
          "xLabel": "Prompt-size target (approximate Haiku tokens; calibration calculation)",
          "yLabel": "Seconds to first text",
          "series": [
            {
              "name": "Claude Haiku 4.5 · Claude Code",
              "points": [
                {
                  "label": "1k",
                  "value": 1.93,
                  "lo": 1.85,
                  "hi": 2.04,
                  "n": 3
                },
                {
                  "label": "16k",
                  "value": 2.27,
                  "lo": 2.22,
                  "hi": 2.47,
                  "n": 3
                },
                {
                  "label": "64k",
                  "value": 2.78,
                  "lo": 2.45,
                  "hi": 2.89,
                  "n": 3
                }
              ]
            },
            {
              "name": "Claude Sonnet 5.5 · Claude Code",
              "points": [
                {
                  "label": "1k",
                  "value": 1.45,
                  "lo": 1.23,
                  "hi": 1.72,
                  "n": 3
                },
                {
                  "label": "16k",
                  "value": 1.78,
                  "lo": 1.64,
                  "hi": 2.11,
                  "n": 3
                },
                {
                  "label": "64k",
                  "value": 3.07,
                  "lo": 1.38,
                  "hi": 3.61,
                  "n": 3
                }
              ]
            },
            {
              "name": "Claude Opus 5.5 · Claude Code",
              "points": [
                {
                  "label": "1k",
                  "value": 1.51,
                  "lo": 1.46,
                  "hi": 2.01,
                  "n": 3
                },
                {
                  "label": "16k",
                  "value": 1.74,
                  "lo": 1.7,
                  "hi": 2.97,
                  "n": 3
                },
                {
                  "label": "64k",
                  "value": 1.79,
                  "lo": 1.72,
                  "hi": 3.72,
                  "n": 3
                }
              ]
            },
            {
              "name": "GPT-6.1 Sol (low) · Codex CLI",
              "points": [
                {
                  "label": "1k",
                  "value": 3.36,
                  "lo": 3.36,
                  "hi": 4.75,
                  "n": 3
                },
                {
                  "label": "16k",
                  "value": 4.02,
                  "lo": 3.3,
                  "hi": 4.28,
                  "n": 3
                },
                {
                  "label": "64k",
                  "value": 3.93,
                  "lo": 3.42,
                  "hi": 4.38,
                  "n": 3
                }
              ]
            }
          ],
          "note": "Each call used a new ledger seed. Cache-read counts stayed within the short-prompt baseline (see the cache table). This does not identify which tokens were cached. Sizes name the text we send; each model’s reported input tokens are in the table and include the CLI’s own prefix. The size calibration subtracts estimated prefixes from probe input counts; these are calculations, not measured prefix counts for each call. Whiskers are a range of calls, not a confidence interval.",
          "whisker": "minmax",
          "sourceIds": [
            "agent-speed-anatomy"
          ]
        },
        {
          "id": "speed-anatomy-total-by-size",
          "title": "Total time per call by prompt size",
          "subtitle": "Median of 3 calls per bar; whiskers = fastest and slowest call",
          "kind": "grouped-bar",
          "unit": "seconds",
          "yLabel": "Seconds, whole call",
          "series": [
            {
              "name": "1k prompt",
              "points": [
                {
                  "label": "Claude Haiku 4.5 · Claude Code",
                  "value": 2.34,
                  "lo": 2.22,
                  "hi": 2.46,
                  "n": 3
                },
                {
                  "label": "Claude Sonnet 5.5 · Claude Code",
                  "value": 1.78,
                  "lo": 1.57,
                  "hi": 2.12,
                  "n": 3
                },
                {
                  "label": "Claude Opus 5.5 · Claude Code",
                  "value": 1.83,
                  "lo": 1.82,
                  "hi": 2.41,
                  "n": 3
                },
                {
                  "label": "GPT-6.1 Sol (low) · Codex CLI",
                  "value": 3.43,
                  "lo": 3.43,
                  "hi": 4.92,
                  "n": 3
                }
              ]
            },
            {
              "name": "16k prompt",
              "points": [
                {
                  "label": "Claude Haiku 4.5 · Claude Code",
                  "value": 2.79,
                  "lo": 2.58,
                  "hi": 2.84,
                  "n": 3
                },
                {
                  "label": "Claude Sonnet 5.5 · Claude Code",
                  "value": 2.1,
                  "lo": 1.98,
                  "hi": 2.48,
                  "n": 3
                },
                {
                  "label": "Claude Opus 5.5 · Claude Code",
                  "value": 2.36,
                  "lo": 2.11,
                  "hi": 3.4,
                  "n": 3
                },
                {
                  "label": "GPT-6.1 Sol (low) · Codex CLI",
                  "value": 4.14,
                  "lo": 3.96,
                  "hi": 4.68,
                  "n": 3
                }
              ]
            },
            {
              "name": "64k prompt",
              "points": [
                {
                  "label": "Claude Haiku 4.5 · Claude Code",
                  "value": 3.13,
                  "lo": 2.84,
                  "hi": 3.28,
                  "n": 3
                },
                {
                  "label": "Claude Sonnet 5.5 · Claude Code",
                  "value": 3.44,
                  "lo": 1.74,
                  "hi": 4.38,
                  "n": 3
                },
                {
                  "label": "Claude Opus 5.5 · Claude Code",
                  "value": 2.35,
                  "lo": 2.26,
                  "hi": 4.29,
                  "n": 3
                },
                {
                  "label": "GPT-6.1 Sol (low) · Codex CLI",
                  "value": 3.96,
                  "lo": 3.47,
                  "hi": 4.44,
                  "n": 3
                }
              ]
            }
          ],
          "note": "Whole call: CLI start-up, first text and the one-line answer. Whiskers are a range of calls, not a confidence interval. Each call used a new ledger.",
          "whisker": "minmax",
          "sourceIds": [
            "agent-speed-anatomy"
          ]
        },
        {
          "id": "speed-anatomy-lookup-correct",
          "title": "Exact lookup answers at the 1k, 16k and 64k prompt-size targets",
          "subtitle": "All sizes together per model; whiskers = 95% Wilson intervals",
          "kind": "dot-range",
          "unit": "rate",
          "polarity": "higher",
          "yLabel": "Lookups answered exactly",
          "series": [
            {
              "name": "Exact answer",
              "points": [
                {
                  "label": "Claude Haiku 4.5 · Claude Code",
                  "value": 1,
                  "lo": 0.7009,
                  "hi": 1,
                  "n": 9
                },
                {
                  "label": "Claude Sonnet 5.5 · Claude Code",
                  "value": 1,
                  "lo": 0.7009,
                  "hi": 1,
                  "n": 9
                },
                {
                  "label": "Claude Opus 5.5 · Claude Code",
                  "value": 0.5556,
                  "lo": 0.2667,
                  "hi": 0.8112,
                  "n": 9
                },
                {
                  "label": "GPT-6.1 Sol (low) · Codex CLI",
                  "value": 1,
                  "lo": 0.7009,
                  "hi": 1,
                  "n": 9
                }
              ]
            }
          ],
          "note": "Whiskers are 95% Wilson intervals. One lookup question per call; a reply with extra words is a format miss, not a pass. With 9 calls per model, a perfect score still has a wide interval.",
          "whisker": "ci95",
          "sourceIds": [
            "agent-speed-anatomy"
          ]
        }
      ],
      "tables": [
        {
          "id": "speed-anatomy-cells",
          "title": "Every speed-anatomy cell, with reasoning tokens",
          "columns": [
            {
              "key": "part",
              "label": "Part",
              "unit": "text"
            },
            {
              "key": "config",
              "label": "Configuration",
              "unit": "text"
            },
            {
              "key": "size",
              "label": "Prompt size",
              "unit": "text"
            },
            {
              "key": "calls",
              "label": "Calls",
              "unit": "count"
            },
            {
              "key": "medianFirst",
              "label": "Median first text (s)",
              "unit": "seconds"
            },
            {
              "key": "rangeFirst",
              "label": "Fastest to slowest (s)",
              "unit": "text"
            },
            {
              "key": "medianAfterReady",
              "label": "Median first text after the CLI was ready (s, calculation)",
              "unit": "seconds"
            },
            {
              "key": "rangeAfterReady",
              "label": "After ready, fastest to slowest (s)",
              "unit": "text"
            },
            {
              "key": "medianTotal",
              "label": "Median total (s)",
              "unit": "seconds"
            },
            {
              "key": "rangeTotal",
              "label": "Total, fastest to slowest (s)",
              "unit": "text"
            },
            {
              "key": "medianInput",
              "label": "Median input tokens reported",
              "unit": "tokens"
            },
            {
              "key": "rangeInput",
              "label": "Input tokens, lowest to highest",
              "unit": "text"
            },
            {
              "key": "medianOut",
              "label": "Median output tokens",
              "unit": "tokens"
            },
            {
              "key": "rangeOut",
              "label": "Output tokens, lowest to highest",
              "unit": "text"
            },
            {
              "key": "medianReasoning",
              "label": "Median reasoning tokens",
              "unit": "tokens"
            },
            {
              "key": "rangeReasoning",
              "label": "Reasoning tokens, lowest to highest",
              "unit": "text"
            },
            {
              "key": "medianVisible",
              "label": "Median visible tokens",
              "unit": "tokens"
            },
            {
              "key": "rangeVisible",
              "label": "Visible tokens, lowest to highest",
              "unit": "text"
            },
            {
              "key": "speedVisible",
              "label": "Visible tokens per second (calculation)",
              "unit": "tokens"
            },
            {
              "key": "rangeSpeed",
              "label": "Visible tokens/s, lowest to highest",
              "unit": "text"
            },
            {
              "key": "speedPlain",
              "label": "All output tokens per second (calculation; inflated when reasoning comes first)",
              "unit": "tokens"
            },
            {
              "key": "rangePlain",
              "label": "All output tokens/s, lowest to highest",
              "unit": "text"
            },
            {
              "key": "charsPerSecond",
              "label": "Characters per second (calculation; exact replies only)",
              "unit": "count"
            },
            {
              "key": "rangeChars",
              "label": "Characters/s, lowest to highest",
              "unit": "text"
            },
            {
              "key": "charsN",
              "label": "Exact replies used for characters/s",
              "unit": "count"
            },
            {
              "key": "correct",
              "label": "Exactly right",
              "unit": "text"
            },
            {
              "key": "lastLine",
              "label": "Not a pass, but the last line is the right number (post-hoc reading)",
              "unit": "count"
            }
          ],
          "rows": [
            {
              "part": "A: 250 numbers in words",
              "config": "Claude Haiku 4.5 · Claude Code",
              "size": null,
              "calls": 4,
              "medianFirst": 4,
              "rangeFirst": "2.84 to 6.38",
              "medianAfterReady": 3.51,
              "medianTotal": 11.9,
              "rangeTotal": "10.91 to 14.31",
              "medianInput": 3658,
              "rangeInput": "3655 to 3658",
              "medianOut": 1780,
              "rangeOut": "1510 to 1997",
              "medianReasoning": 413,
              "rangeReasoning": "255 to 614",
              "medianVisible": 1210,
              "rangeVisible": "1210 to 1742",
              "speedVisible": 153.2,
              "rangeSpeed": "152.6 to 216.1",
              "rangeChars": "546 to 548",
              "charsN": 3,
              "rangeAfterReady": "2.36 to 5.78",
              "speedPlain": 225,
              "rangePlain": "190.9 to 247.7",
              "charsPerSecond": 547,
              "correct": "3/4",
              "lastLine": null
            },
            {
              "part": "A: 250 numbers in words",
              "config": "Claude Sonnet 5.5 · Claude Code",
              "size": null,
              "calls": 4,
              "medianFirst": 1.96,
              "rangeFirst": "0.88 to 4.09",
              "medianAfterReady": 1.51,
              "medianTotal": 10.33,
              "rangeTotal": "9.31 to 12.42",
              "medianInput": 1913,
              "rangeInput": "1912 to 1914",
              "medianOut": 2024,
              "rangeOut": "1941 to 2339",
              "medianReasoning": 83,
              "rangeReasoning": "0 to 398",
              "medianVisible": 1941,
              "rangeVisible": "1941 to 1941",
              "speedVisible": 231.7,
              "rangeSpeed": "230.3 to 233",
              "rangeChars": "513 to 519",
              "charsN": 4,
              "rangeAfterReady": "0.42 to 3.64",
              "speedPlain": 241.6,
              "rangePlain": "230.3 to 280.8",
              "charsPerSecond": 517,
              "correct": "4/4",
              "lastLine": null
            },
            {
              "part": "A: 250 numbers in words",
              "config": "Claude Opus 5.5 · Claude Code",
              "size": null,
              "calls": 4,
              "medianFirst": 1.97,
              "rangeFirst": "1.7 to 2.35",
              "medianAfterReady": 1.52,
              "medianTotal": 14.43,
              "rangeTotal": "14.25 to 14.8",
              "medianInput": 1909,
              "rangeInput": "1908 to 1909",
              "medianOut": 1990,
              "rangeOut": "1985 to 1994",
              "medianReasoning": 49,
              "rangeReasoning": "44 to 53",
              "medianVisible": 1941,
              "rangeVisible": "1941 to 1941",
              "speedVisible": 155.5,
              "rangeSpeed": "154.6 to 156.4",
              "rangeChars": "345 to 349",
              "charsN": 4,
              "rangeAfterReady": "1.24 to 1.87",
              "speedPlain": 159.4,
              "rangePlain": "158.7 to 160.3",
              "charsPerSecond": 347,
              "correct": "4/4",
              "lastLine": null
            },
            {
              "part": "A: 250 numbers in words",
              "config": "Claude Fable 5.1 · Claude Code",
              "size": null,
              "calls": 4,
              "medianFirst": 4.43,
              "rangeFirst": "2.27 to 4.64",
              "medianAfterReady": 3.95,
              "medianTotal": 19.81,
              "rangeTotal": "18.32 to 20.31",
              "medianInput": 3768,
              "rangeInput": "3766 to 3768",
              "medianOut": 2092,
              "rangeOut": "1941 to 2156",
              "medianReasoning": 151,
              "rangeReasoning": "0 to 215",
              "medianVisible": 1941,
              "rangeVisible": "1941 to 1941",
              "speedVisible": 122.6,
              "rangeSpeed": "120.9 to 131.4",
              "rangeChars": "270 to 293",
              "charsN": 4,
              "rangeAfterReady": "1.65 to 4.21",
              "speedPlain": 130.9,
              "rangePlain": "123.2 to 145.9",
              "charsPerSecond": 273,
              "correct": "4/4",
              "lastLine": null
            },
            {
              "part": "A: 250 numbers in words",
              "config": "GPT-6.1 Sol (low) · Codex CLI",
              "size": null,
              "calls": 4,
              "medianFirst": 3.52,
              "rangeFirst": "2.75 to 4.42",
              "medianAfterReady": 3.07,
              "medianTotal": 17.51,
              "rangeTotal": "16.36 to 17.75",
              "medianInput": 12028,
              "rangeInput": "12026 to 12028",
              "medianOut": 1066,
              "rangeOut": "1066 to 1066",
              "medianReasoning": 0,
              "rangeReasoning": "0 to 0",
              "medianVisible": 1066,
              "rangeVisible": "1066 to 1066",
              "speedVisible": 79.6,
              "rangeSpeed": "71.6 to 80.5",
              "rangeChars": "291 to 327",
              "charsN": 4,
              "rangeAfterReady": "2.34 to 4.01",
              "speedPlain": 79.6,
              "rangePlain": "71.6 to 80.5",
              "charsPerSecond": 323,
              "correct": "4/4",
              "lastLine": null
            },
            {
              "part": "A: 250 numbers in words",
              "config": "GPT-6 Luna (low) · Codex CLI",
              "size": null,
              "calls": 4,
              "medianFirst": 3.3,
              "rangeFirst": "3.19 to 3.47",
              "medianAfterReady": 2.51,
              "medianTotal": 15.5,
              "rangeTotal": "7.39 to 22.69",
              "medianInput": 11335,
              "rangeInput": "11332 to 11336",
              "medianOut": 1066,
              "rangeOut": "1066 to 1066",
              "medianReasoning": 0,
              "rangeReasoning": "0 to 0",
              "medianVisible": 1066,
              "rangeVisible": "1066 to 1066",
              "speedVisible": 129.1,
              "rangeSpeed": "55.5 to 259.1",
              "rangeChars": "225 to 1052",
              "charsN": 4,
              "rangeAfterReady": "2.14 to 2.88",
              "speedPlain": 129.1,
              "rangePlain": "55.5 to 259.1",
              "charsPerSecond": 524,
              "correct": "4/4",
              "lastLine": null
            },
            {
              "part": "B: ledger lookup",
              "config": "Claude Haiku 4.5 · Claude Code",
              "size": "1k",
              "calls": 3,
              "medianFirst": 1.93,
              "rangeFirst": "1.85 to 2.04",
              "medianAfterReady": 1.41,
              "medianTotal": 2.34,
              "rangeTotal": "2.22 to 2.46",
              "medianInput": 4592,
              "rangeInput": "4580 to 4610",
              "medianOut": 130,
              "rangeOut": "126 to 133",
              "medianReasoning": 123,
              "rangeReasoning": "119 to 126",
              "medianVisible": 7,
              "rangeVisible": "7 to 7",
              "speedVisible": 16.9,
              "rangeSpeed": "16.5 to 18.9",
              "rangeChars": "7 to 8",
              "charsN": 3,
              "rangeAfterReady": "1.38 to 1.43",
              "speedPlain": 314,
              "rangePlain": "296.5 to 358.5",
              "charsPerSecond": 7,
              "correct": "3/3",
              "lastLine": 0
            },
            {
              "part": "B: ledger lookup",
              "config": "Claude Haiku 4.5 · Claude Code",
              "size": "16k",
              "calls": 3,
              "medianFirst": 2.27,
              "rangeFirst": "2.22 to 2.47",
              "medianAfterReady": 1.76,
              "medianTotal": 2.79,
              "rangeTotal": "2.58 to 2.84",
              "medianInput": 19618,
              "rangeInput": "19604 to 19632",
              "medianOut": 136,
              "rangeOut": "129 to 139",
              "medianReasoning": 129,
              "rangeReasoning": "122 to 132",
              "medianVisible": 7,
              "rangeVisible": "7 to 7",
              "speedVisible": 19.1,
              "rangeSpeed": "13.5 to 19.4",
              "rangeChars": "6 to 8",
              "charsN": 3,
              "rangeAfterReady": "1.64 to 2.01",
              "speedPlain": 358.3,
              "rangePlain": "268.3 to 371.6",
              "charsPerSecond": 8,
              "correct": "3/3",
              "lastLine": 0
            },
            {
              "part": "B: ledger lookup",
              "config": "Claude Haiku 4.5 · Claude Code",
              "size": "64k",
              "calls": 3,
              "medianFirst": 2.78,
              "rangeFirst": "2.45 to 2.89",
              "medianAfterReady": 2.32,
              "medianTotal": 3.13,
              "rangeTotal": "2.84 to 3.28",
              "medianInput": 67605,
              "rangeInput": "67578 to 67639",
              "medianOut": 157,
              "rangeOut": "132 to 167",
              "medianReasoning": 150,
              "rangeReasoning": "125 to 160",
              "medianVisible": 7,
              "rangeVisible": "7 to 7",
              "speedVisible": 18.2,
              "rangeSpeed": "17.9 to 19.9",
              "rangeChars": "8 to 9",
              "charsN": 3,
              "rangeAfterReady": "1.94 to 2.42",
              "speedPlain": 401.5,
              "rangePlain": "343.8 to 475.8",
              "charsPerSecond": 8,
              "correct": "3/3",
              "lastLine": 0
            },
            {
              "part": "B: ledger lookup",
              "config": "Claude Sonnet 5.5 · Claude Code",
              "size": "1k",
              "calls": 3,
              "medianFirst": 1.45,
              "rangeFirst": "1.23 to 1.72",
              "medianAfterReady": 0.98,
              "medianTotal": 1.78,
              "rangeTotal": "1.57 to 2.12",
              "medianInput": 3051,
              "rangeInput": "3049 to 3057",
              "medianOut": 3,
              "rangeOut": "3 to 3",
              "medianReasoning": 0,
              "rangeReasoning": "0 to 0",
              "medianVisible": 3,
              "rangeVisible": "3 to 3",
              "speedVisible": 8.7,
              "rangeSpeed": "7.4 to 9.1",
              "rangeChars": "7 to 9",
              "charsN": 3,
              "rangeAfterReady": "0.8 to 0.98",
              "speedPlain": 8.7,
              "rangePlain": "7.4 to 9.1",
              "charsPerSecond": 9,
              "correct": "3/3",
              "lastLine": 0
            },
            {
              "part": "B: ledger lookup",
              "config": "Claude Sonnet 5.5 · Claude Code",
              "size": "16k",
              "calls": 3,
              "medianFirst": 1.78,
              "rangeFirst": "1.64 to 2.11",
              "medianAfterReady": 1.28,
              "medianTotal": 2.1,
              "rangeTotal": "1.98 to 2.48",
              "medianInput": 21080,
              "rangeInput": "21050 to 21099",
              "medianOut": 3,
              "rangeOut": "3 to 3",
              "medianReasoning": 0,
              "rangeReasoning": "0 to 0",
              "medianVisible": 3,
              "rangeVisible": "3 to 3",
              "speedVisible": 8.9,
              "rangeSpeed": "8.2 to 9.2",
              "rangeChars": "5 to 9",
              "charsN": 3,
              "rangeAfterReady": "1.15 to 1.32",
              "speedPlain": 8.9,
              "rangePlain": "8.2 to 9.2",
              "charsPerSecond": 9,
              "correct": "3/3",
              "lastLine": 0
            },
            {
              "part": "B: ledger lookup",
              "config": "Claude Sonnet 5.5 · Claude Code",
              "size": "64k",
              "calls": 3,
              "medianFirst": 3.07,
              "rangeFirst": "1.38 to 3.61",
              "medianAfterReady": 2.36,
              "medianTotal": 3.44,
              "rangeTotal": "1.74 to 4.38",
              "medianInput": 78596,
              "rangeInput": "78577 to 78745",
              "medianOut": 3,
              "rangeOut": "3 to 3",
              "medianReasoning": 0,
              "rangeReasoning": "0 to 0",
              "medianVisible": 3,
              "rangeVisible": "3 to 3",
              "speedVisible": 8,
              "rangeSpeed": "3.9 to 8.3",
              "rangeChars": "4 to 8",
              "charsN": 3,
              "rangeAfterReady": "0.94 to 3.13",
              "speedPlain": 8,
              "rangePlain": "3.9 to 8.3",
              "charsPerSecond": 8,
              "correct": "3/3",
              "lastLine": 0
            },
            {
              "part": "B: ledger lookup",
              "config": "Claude Opus 5.5 · Claude Code",
              "size": "1k",
              "calls": 3,
              "medianFirst": 1.51,
              "rangeFirst": "1.46 to 2.01",
              "medianAfterReady": 1.02,
              "medianTotal": 1.83,
              "rangeTotal": "1.82 to 2.41",
              "medianInput": 3040,
              "rangeInput": "3040 to 3044",
              "medianOut": 3,
              "rangeOut": "3 to 3",
              "medianReasoning": 0,
              "rangeReasoning": "0 to 0",
              "medianVisible": 3,
              "rangeVisible": "3 to 3",
              "speedVisible": 8.1,
              "rangeSpeed": "7.5 to 9.4",
              "rangeChars": "8 to 9",
              "charsN": 3,
              "rangeAfterReady": "1 to 1.36",
              "speedPlain": 8.1,
              "rangePlain": "7.5 to 9.4",
              "charsPerSecond": 8,
              "correct": "3/3",
              "lastLine": 0
            },
            {
              "part": "B: ledger lookup",
              "config": "Claude Opus 5.5 · Claude Code",
              "size": "16k",
              "calls": 3,
              "medianFirst": 1.74,
              "rangeFirst": "1.7 to 2.97",
              "medianAfterReady": 1.28,
              "medianTotal": 2.36,
              "rangeTotal": "2.11 to 3.4",
              "medianInput": 21074,
              "rangeInput": "20998 to 21089",
              "medianOut": 3,
              "rangeOut": "3 to 36",
              "medianReasoning": 0,
              "rangeReasoning": "0 to 0",
              "medianVisible": 3,
              "rangeVisible": "3 to 36",
              "speedVisible": 7.3,
              "rangeSpeed": "7 to 58.3",
              "rangeChars": "7 to 7",
              "charsN": 2,
              "rangeAfterReady": "1.19 to 2.53",
              "speedPlain": 7.3,
              "rangePlain": "7 to 58.3",
              "charsPerSecond": 7,
              "correct": "2/3",
              "lastLine": 1
            },
            {
              "part": "B: ledger lookup",
              "config": "Claude Opus 5.5 · Claude Code",
              "size": "64k",
              "calls": 3,
              "medianFirst": 1.79,
              "rangeFirst": "1.72 to 3.72",
              "medianAfterReady": 1.32,
              "medianTotal": 2.35,
              "rangeTotal": "2.26 to 4.29",
              "medianInput": 78674,
              "rangeInput": "78633 to 78684",
              "medianOut": 33,
              "rangeOut": "31 to 34",
              "medianReasoning": 0,
              "rangeReasoning": "0 to 0",
              "medianVisible": 33,
              "rangeVisible": "31 to 34",
              "speedVisible": 57.6,
              "rangeSpeed": "57.5 to 60.4",
              "rangeChars": null,
              "charsN": 0,
              "rangeAfterReady": "1.29 to 3.25",
              "speedPlain": 57.6,
              "rangePlain": "57.5 to 60.4",
              "charsPerSecond": null,
              "correct": "0/3",
              "lastLine": 3
            },
            {
              "part": "B: ledger lookup",
              "config": "GPT-6.1 Sol (low) · Codex CLI",
              "size": "1k",
              "calls": 3,
              "medianFirst": 3.36,
              "rangeFirst": "3.36 to 4.75",
              "medianAfterReady": 3.02,
              "medianTotal": 3.43,
              "rangeTotal": "3.43 to 4.92",
              "medianInput": 12790,
              "rangeInput": "12787 to 12797",
              "medianOut": 5,
              "rangeOut": "5 to 5",
              "medianReasoning": 0,
              "rangeReasoning": "0 to 0",
              "medianVisible": 5,
              "rangeVisible": "5 to 5",
              "speedVisible": 67.6,
              "rangeSpeed": "28.4 to 68.5",
              "rangeChars": "17 to 41",
              "charsN": 3,
              "rangeAfterReady": "2.76 to 3.29",
              "speedPlain": 67.6,
              "rangePlain": "28.4 to 68.5",
              "charsPerSecond": 41,
              "correct": "3/3",
              "lastLine": 0
            },
            {
              "part": "B: ledger lookup",
              "config": "GPT-6.1 Sol (low) · Codex CLI",
              "size": "16k",
              "calls": 3,
              "medianFirst": 4.02,
              "rangeFirst": "3.3 to 4.28",
              "medianAfterReady": 3.68,
              "medianTotal": 4.14,
              "rangeTotal": "3.96 to 4.68",
              "medianInput": 25038,
              "rangeInput": "25030 to 25045",
              "medianOut": 5,
              "rangeOut": "5 to 5",
              "medianReasoning": 0,
              "rangeReasoning": "0 to 0",
              "medianVisible": 5,
              "rangeVisible": "5 to 5",
              "speedVisible": 12.5,
              "rangeSpeed": "7.5 to 39.4",
              "rangeChars": "5 to 24",
              "charsN": 3,
              "rangeAfterReady": "2.88 to 3.95",
              "speedPlain": 12.5,
              "rangePlain": "7.5 to 39.4",
              "charsPerSecond": 7,
              "correct": "3/3",
              "lastLine": 0
            },
            {
              "part": "B: ledger lookup",
              "config": "GPT-6.1 Sol (low) · Codex CLI",
              "size": "64k",
              "calls": 3,
              "medianFirst": 3.93,
              "rangeFirst": "3.42 to 4.38",
              "medianAfterReady": 3.59,
              "medianTotal": 3.96,
              "rangeTotal": "3.47 to 4.44",
              "medianInput": 64172,
              "rangeInput": "64118 to 64198",
              "medianOut": 5,
              "rangeOut": "5 to 5",
              "medianReasoning": 0,
              "rangeReasoning": "0 to 0",
              "medianVisible": 5,
              "rangeVisible": "5 to 5",
              "speedVisible": 86.2,
              "rangeSpeed": "73.5 to 166.7",
              "rangeChars": "44 to 100",
              "charsN": 3,
              "rangeAfterReady": "3.02 to 4.06",
              "speedPlain": 86.2,
              "rangePlain": "73.5 to 166.7",
              "charsPerSecond": 52,
              "correct": "3/3",
              "lastLine": 0
            }
          ]
        },
        {
          "id": "speed-anatomy-cache-reads",
          "title": "Cache reads and writes per call; share = cache-read tokens ÷ reported input tokens (calculation)",
          "columns": [
            {
              "key": "config",
              "label": "Configuration",
              "unit": "text"
            },
            {
              "key": "size",
              "label": "Prompt size",
              "unit": "text"
            },
            {
              "key": "calls",
              "label": "Calls",
              "unit": "count"
            },
            {
              "key": "medianInput",
              "label": "Median input tokens reported",
              "unit": "tokens"
            },
            {
              "key": "rangeInput",
              "label": "Input tokens, lowest to highest",
              "unit": "text"
            },
            {
              "key": "medianRead",
              "label": "Median cache-read tokens",
              "unit": "tokens"
            },
            {
              "key": "rangeRead",
              "label": "Read tokens, lowest to highest",
              "unit": "text"
            },
            {
              "key": "maxRead",
              "label": "Most cache-read tokens in one call",
              "unit": "tokens"
            },
            {
              "key": "medianWrite",
              "label": "Median cache-write tokens",
              "unit": "tokens"
            },
            {
              "key": "rangeWrite",
              "label": "Write tokens, lowest to highest",
              "unit": "text"
            },
            {
              "key": "readShare",
              "label": "Median cache-read share of input (calculation)",
              "unit": "rate"
            },
            {
              "key": "rangeReadShare",
              "label": "Cache-read share, lowest to highest (calculation)",
              "unit": "text"
            },
            {
              "key": "readShareN",
              "label": "Calls used for cache-read share",
              "unit": "count"
            }
          ],
          "rows": [
            {
              "config": "Claude Haiku 4.5 · Claude Code",
              "size": "1k",
              "calls": 3,
              "medianInput": 4592,
              "rangeInput": "4580 to 4610",
              "medianRead": 0,
              "rangeRead": "0 to 0",
              "maxRead": 0,
              "medianWrite": 4582,
              "rangeWrite": "4570 to 4600",
              "readShare": 0,
              "rangeReadShare": "0.0% to 0.0%",
              "readShareN": 3
            },
            {
              "config": "Claude Haiku 4.5 · Claude Code",
              "size": "16k",
              "calls": 3,
              "medianInput": 19618,
              "rangeInput": "19604 to 19632",
              "medianRead": 0,
              "rangeRead": "0 to 0",
              "maxRead": 0,
              "medianWrite": 19608,
              "rangeWrite": "19594 to 19622",
              "readShare": 0,
              "rangeReadShare": "0.0% to 0.0%",
              "readShareN": 3
            },
            {
              "config": "Claude Haiku 4.5 · Claude Code",
              "size": "64k",
              "calls": 3,
              "medianInput": 67605,
              "rangeInput": "67578 to 67639",
              "medianRead": 0,
              "rangeRead": "0 to 0",
              "maxRead": 0,
              "medianWrite": 67595,
              "rangeWrite": "67568 to 67629",
              "readShare": 0,
              "rangeReadShare": "0.0% to 0.0%",
              "readShareN": 3
            },
            {
              "config": "Claude Sonnet 5.5 · Claude Code",
              "size": "1k",
              "calls": 3,
              "medianInput": 3051,
              "rangeInput": "3049 to 3057",
              "medianRead": 1463,
              "rangeRead": "1463 to 1463",
              "maxRead": 1463,
              "medianWrite": 1586,
              "rangeWrite": "1584 to 1592",
              "readShare": 0.48,
              "rangeReadShare": "47.9% to 48.0%",
              "readShareN": 3
            },
            {
              "config": "Claude Sonnet 5.5 · Claude Code",
              "size": "16k",
              "calls": 3,
              "medianInput": 21080,
              "rangeInput": "21050 to 21099",
              "medianRead": 1463,
              "rangeRead": "1463 to 1463",
              "maxRead": 1463,
              "medianWrite": 19615,
              "rangeWrite": "19585 to 19634",
              "readShare": 0.069,
              "rangeReadShare": "6.9% to 7.0%",
              "readShareN": 3
            },
            {
              "config": "Claude Sonnet 5.5 · Claude Code",
              "size": "64k",
              "calls": 3,
              "medianInput": 78596,
              "rangeInput": "78577 to 78745",
              "medianRead": 1463,
              "rangeRead": "1463 to 1463",
              "maxRead": 1463,
              "medianWrite": 77131,
              "rangeWrite": "77112 to 77280",
              "readShare": 0.019,
              "rangeReadShare": "1.9% to 1.9%",
              "readShareN": 3
            },
            {
              "config": "Claude Opus 5.5 · Claude Code",
              "size": "1k",
              "calls": 3,
              "medianInput": 3040,
              "rangeInput": "3040 to 3044",
              "medianRead": 1463,
              "rangeRead": "1463 to 1463",
              "maxRead": 1463,
              "medianWrite": 1575,
              "rangeWrite": "1575 to 1579",
              "readShare": 0.481,
              "rangeReadShare": "48.1% to 48.1%",
              "readShareN": 3
            },
            {
              "config": "Claude Opus 5.5 · Claude Code",
              "size": "16k",
              "calls": 3,
              "medianInput": 21074,
              "rangeInput": "20998 to 21089",
              "medianRead": 1463,
              "rangeRead": "1463 to 1463",
              "maxRead": 1463,
              "medianWrite": 19609,
              "rangeWrite": "19533 to 19624",
              "readShare": 0.069,
              "rangeReadShare": "6.9% to 7.0%",
              "readShareN": 3
            },
            {
              "config": "Claude Opus 5.5 · Claude Code",
              "size": "64k",
              "calls": 3,
              "medianInput": 78674,
              "rangeInput": "78633 to 78684",
              "medianRead": 1463,
              "rangeRead": "1463 to 1463",
              "maxRead": 1463,
              "medianWrite": 77209,
              "rangeWrite": "77168 to 77219",
              "readShare": 0.019,
              "rangeReadShare": "1.9% to 1.9%",
              "readShareN": 3
            },
            {
              "config": "GPT-6.1 Sol (low) · Codex CLI",
              "size": "1k",
              "calls": 3,
              "medianInput": 12790,
              "rangeInput": "12787 to 12797",
              "medianRead": 8960,
              "rangeRead": "0 to 8960",
              "maxRead": 8960,
              "medianWrite": 0,
              "rangeWrite": "0 to 0",
              "readShare": 0.7,
              "rangeReadShare": "0.0% to 70.1%",
              "readShareN": 3
            },
            {
              "config": "GPT-6.1 Sol (low) · Codex CLI",
              "size": "16k",
              "calls": 3,
              "medianInput": 25038,
              "rangeInput": "25030 to 25045",
              "medianRead": 8960,
              "rangeRead": "8960 to 8960",
              "maxRead": 8960,
              "medianWrite": 0,
              "rangeWrite": "0 to 0",
              "readShare": 0.358,
              "rangeReadShare": "35.8% to 35.8%",
              "readShareN": 3
            },
            {
              "config": "GPT-6.1 Sol (low) · Codex CLI",
              "size": "64k",
              "calls": 3,
              "medianInput": 64172,
              "rangeInput": "64118 to 64198",
              "medianRead": 8960,
              "rangeRead": "8960 to 8960",
              "maxRead": 8960,
              "medianWrite": 0,
              "rangeWrite": "0 to 0",
              "readShare": 0.14,
              "rangeReadShare": "14.0% to 14.0%",
              "readShareN": 3
            }
          ]
        }
      ],
      "related": [
        "cli-model-latency-tokens",
        "model-head-to-head",
        "routing-overhead"
      ]
    },
    {
      "slug": "haiku-retry-or-escalate",
      "title": "Is Claude Haiku cheaper? Retry and escalate, calculated on real receipts",
      "seoTitle": "Haiku retry and escalate: cost per correct answer",
      "description": "Haiku 4.5 first, then retry or escalate to Sonnet 5.5? A calculation on 64 recorded calls across 8 hard tasks: cost and time per correct answer.",
      "question": "On the 8 hard tasks, what does one correct answer cost, and how long does it take, if you try Claude Haiku 4.5 first and retry or escalate, against Claude Sonnet 5.5 every time?",
      "answer": "The calculated Haiku-first policies cost more on these 8 hard tasks. These are scenarios from recorded calls, not a measured policy ranking. Sonnet every time costs $0.0132 per correct answer and takes 8.2 s (24/24 passes; 95% interval 86% to 100%). Haiku passes 11/24 calls (95% interval 28% to 65%); Haiku once costs $0.0677 per correct answer. Haiku with one retry, then Sonnet, costs $0.0566 and takes 71.5 s per correct answer. The tables show the other policies and sensitivity ranges; the method states the assumptions.",
      "date": "2026-10-07",
      "updated": "2026-10-07",
      "tags": [
        "thought-experiment",
        "claude-haiku",
        "claude-sonnet",
        "llm-pricing",
        "cost-per-correct-answer",
        "retry",
        "escalation",
        "hard-tasks"
      ],
      "method": [
        "This study makes no new model call and adds no new raw file. It reuses recorded calls from the hard head-to-head (Haiku 4.5 and Sonnet 5.5 at default effort, 3 calls per task) and the effort ladder (Sonnet 5.5 at low effort, 2 calls per task). All calls ran in Claude Code on the same 8 tasks with the same strict validators.",
        "The source controls ran before counted inference. Each controls receipt records 8 passing references, 26 rejected wrong answers and 8 detected wrapped format misses. The hard-set protocol outcome says 27 wrong answers; the controls receipt supports 26. Both Claude batches completed their declared call caps: 120 hard-set calls and 80 ladder calls, with no trims or errors. They used a 300 s timeout and a 16,000-output-token setting. The selected calls stayed below that setting. No Claude probe is recorded in these source folders; the hard-set Codex probe is outside this calculation.",
        "For each task and model, we take the pass rate p. It is strict passes ÷ calls, with its 95% Wilson interval. A format miss is not a pass.",
        "These are median-input scenarios. A median is not an expected mean. Cost and time can depend on whether a call fails. We do not model that link. We take the cost of one call as the median list-price cost of that task’s calls. List price is reported tokens × price. Cache reads and one-hour cache writes are priced as in the hard head-to-head.",
        "We take the time of one call as the median total time of that task’s calls. It includes CLI start-up.",
        "A policy is a list of tries. A try runs only when every earlier try failed. The policies are: Sonnet every time, Haiku once, Haiku with one retry then Sonnet, Haiku with up to 3 tries then Sonnet, and Sonnet at low effort once.",
        "Per task, let q = 1 − p(Haiku). For Haiku with one retry then Sonnet: cost = C(Haiku) × (1 + q) + C(Sonnet) × q². Time = T(Haiku) × (1 + q) + T(Sonnet) × q². Success = 1 − q² × (1 − p(Sonnet)).",
        "For Haiku with up to 3 tries then Sonnet: cost = C(Haiku) × (1 + q + q²) + C(Sonnet) × q³. Success = 1 − q³ × (1 − p(Sonnet)). Time uses the same weights.",
        "Over the 8 tasks, each task is equally likely. Cost per correct answer = total expected cost ÷ total expected correct answers. We compute time the same way, with the tries in sequence.",
        "We assume three things. A validator detects a failed answer, and running it is free. Tries are independent at each task’s observed rate. A failed last Sonnet try is a miss. Every figure that follows is a calculation.",
        "For the sensitivity, we set Haiku’s pass rate on all tasks at once to the lower end of its 95% Wilson interval, then to the upper end. A fourth setting counts Haiku’s format misses as passes. The result is a sensitivity range, not a confidence interval.",
        "A joint extreme moves both models at once: Haiku’s pass rate to the upper end and Sonnet’s to the lower end of each task’s 95% Wilson interval. It is more extreme than a 95% interval on the total, and it is not a forecast.",
        "For the break-even, we scale Haiku’s call cost (or time) by a factor f and keep all pass rates. We solve for the f at which Haiku with one retry then Sonnet equals Sonnet every time: f = (Σ C(Sonnet) − Σ C(Sonnet) × q²) ÷ Σ C(Haiku) × (1 + q). This is arithmetic, not a forecast."
      ],
      "caveats": [
        "Advance registration is not verified. The hard-set protocol file birth time is 2026-10-06 04:03:37 UTC; its first counted call started at 03:23:59 UTC. The ladder protocol file birth time is 14:35:11 UTC; its first Claude call started at 14:21:50 UTC. Both files claim advance declaration, but the available file times do not support that claim. Copying could explain the times; we cannot establish it.",
        "Wilson intervals treat calls as independent. Repeated calls on eight fixed tasks do not give a population interval for coding work. No task-level paired test or measured retry policy supports a general ranking.",
        "This is a calculation, not a run. No policy ran. Costs are list-price calculations from reported tokens. The calls ran on a flat subscription, so no invoice backs them.",
        "Retry independence is untested. Calls repeat the same eight prompts. Failures can repeat, and escalation after a Haiku failure may differ from a fresh Sonnet call. The sensitivity does not cover these links.",
        "Samples are small: 3 calls per task for Haiku and for Sonnet, and 2 for Sonnet at low effort. A per-task pass rate has a wide interval: 0 of 3 allows up to 56%, and 3 of 3 allows as low as 44%. A median of a few calls has no interval. The sensitivity moves all 8 tasks to a bound at once. That is more extreme than a 95% interval on the total.",
        "Sonnet 5.5 passed 24/24, and the model treats it as always right (95% interval 86% to 100%). So every policy that ends in a Sonnet try shows 100% success. A perfect sample does not prove perfect success on future calls. This is the ceiling effect of the task set.",
        "Haiku ran at the CLI’s default effort: the effort flag was not passed, so the CLI chose. It wrote a median 5,064 output tokens per call against 1,050 for Sonnet. Output tokens enter the price calculation. The run does not show what caused the timing gap. We did not test Haiku with thinking off. The break-even stats show how far its calls would have to fall.",
        "Strict grading counts a reply in a code fence as a failure. 5 of Haiku’s 13 failed calls were such format misses. With them counted as passes (lenient reading), Haiku once costs $0.0466 per correct answer, and Haiku with one retry then Sonnet costs $0.0459. Both stay above Sonnet every time.",
        "This study uses each task’s median call for cost and time. The hard head-to-head divides the total cost of all calls by the passes instead. So its figures differ a little: Sonnet $0.0143 and Haiku $0.0672 per strict pass, against $0.0132 and $0.0677 here.",
        "This is a hand-built task set, tuned after an earlier set hit a ceiling. It is not a random sample of coding work. This study covers the eight hard tasks only. Haiku may do better on easy tasks, where pass rate hits a ceiling. We did not recalculate the five-task head-to-head here.",
        "Haiku and Sonnet at default effort ran in one batch on 2026-10-06 (UTC). Sonnet at low effort ran about 10 hours later in the effort ladder. All calls used one host and one network. We did not control for other load on the host, so wall times may carry some contention. Times include CLI start-up and the CLI’s own system prompt.",
        "The sensitivity moves Haiku’s pass rate only, and the Sonnet baseline rests on 24/24. In a joint extreme (Haiku at the upper and Sonnet at the lower end of its Wilson interval on every task), the closest Haiku-first policy costs 1.3 times as much, and the quickest takes 2.8 times as long, as Sonnet every time (calculation). That setting has Sonnet succeed on only 44% of the tasks, which no run showed."
      ],
      "sourceIds": [
        "calc-repricing",
        "agent-provider-h2h-hard",
        "agent-effort-ladder",
        "price-anthropic"
      ],
      "stats": [
        {
          "id": "retry-escalate-haiku-passes",
          "label": "Strict passes, Claude Haiku 4.5 · Claude Code, eight hard tasks",
          "value": 0.4583,
          "unit": "rate",
          "display": "46% (11/24)",
          "n": 24,
          "ci": [
            0.2789,
            0.6493
          ]
        },
        {
          "id": "retry-escalate-sonnet-passes",
          "label": "Strict passes, Claude Sonnet 5.5 · Claude Code, eight hard tasks",
          "value": 1,
          "unit": "rate",
          "display": "100% (24/24)",
          "n": 24,
          "ci": [
            0.862,
            1
          ]
        },
        {
          "id": "retry-escalate-sonnet-low-passes",
          "label": "Strict passes, Claude Sonnet 5.5 (low) · Claude Code, eight hard tasks",
          "value": 1,
          "unit": "rate",
          "display": "100% (16/16)",
          "n": 16,
          "ci": [
            0.8064,
            1
          ]
        },
        {
          "id": "retry-escalate-baseline-cost",
          "label": "Cost per correct answer, Sonnet 5.5 every time (calculation)",
          "value": 0.01322,
          "unit": "usd",
          "display": "$0.0132",
          "n": 24
        },
        {
          "id": "retry-escalate-baseline-time",
          "label": "Wall time per correct answer, Sonnet 5.5 every time (calculation)",
          "value": 8.24,
          "unit": "seconds",
          "display": "8.2 s",
          "n": 24
        },
        {
          "id": "retry-escalate-haiku-once-cost",
          "label": "Cost per correct answer, Haiku once (calculation)",
          "value": 0.06772,
          "unit": "usd",
          "display": "$0.0677",
          "n": 24
        },
        {
          "id": "retry-escalate-haiku-once-time",
          "label": "Wall time per correct answer, Haiku once (calculation)",
          "value": 91.47,
          "unit": "seconds",
          "display": "91.5 s",
          "n": 24
        },
        {
          "id": "retry-escalate-retry-cost",
          "label": "Cost per correct answer, Haiku with one retry then Sonnet (calculation)",
          "value": 0.05664,
          "unit": "usd",
          "display": "$0.0566",
          "n": 48
        },
        {
          "id": "retry-escalate-retry-time",
          "label": "Wall time per correct answer, Haiku with one retry then Sonnet (calculation)",
          "value": 71.54,
          "unit": "seconds",
          "display": "71.5 s",
          "n": 48
        },
        {
          "id": "retry-escalate-retry-cost-ratio",
          "label": "Haiku with one retry then Sonnet vs Sonnet every time: cost per correct answer (calculation)",
          "value": 4.28,
          "unit": "ratio",
          "display": "4.3x",
          "n": 48
        },
        {
          "id": "retry-escalate-retry-time-ratio",
          "label": "Haiku with one retry then Sonnet vs Sonnet every time: wall time per correct answer (calculation)",
          "value": 8.68,
          "unit": "ratio",
          "display": "8.7x",
          "n": 48
        },
        {
          "id": "retry-escalate-three-tries-cost-ratio",
          "label": "Haiku up to 3 tries then Sonnet vs Sonnet every time: cost per correct answer (calculation)",
          "value": 5.44,
          "unit": "ratio",
          "display": "5.4x",
          "n": 48
        },
        {
          "id": "retry-escalate-sonnet-low-cost",
          "label": "Cost per correct answer, Sonnet 5.5 low effort once (calculation)",
          "value": 0.01219,
          "unit": "usd",
          "display": "$0.0122",
          "n": 16
        },
        {
          "id": "retry-escalate-sonnet-low-time",
          "label": "Wall time per correct answer, Sonnet 5.5 low effort once (calculation)",
          "value": 7.07,
          "unit": "seconds",
          "display": "7.1 s",
          "n": 16
        },
        {
          "id": "retry-escalate-haiku-dearer-tasks",
          "label": "Tasks where Haiku’s median call cost more than Sonnet’s (calculation)",
          "value": 8,
          "unit": "count",
          "display": "8 of 8",
          "n": 8
        },
        {
          "id": "retry-escalate-haiku-slower-tasks",
          "label": "Tasks where Haiku’s median call took longer than Sonnet’s",
          "value": 8,
          "unit": "count",
          "display": "8 of 8",
          "n": 8
        },
        {
          "id": "retry-escalate-haiku-output-tokens",
          "label": "Median output tokens per call, Haiku 4.5 (Claude Code)",
          "value": 5064,
          "unit": "tokens",
          "display": "5,064",
          "n": 24,
          "note": "Call range 1,899 to 9,321 tokens; not a confidence interval."
        },
        {
          "id": "retry-escalate-sonnet-output-tokens",
          "label": "Median output tokens per call, Sonnet 5.5 (Claude Code)",
          "value": 1050,
          "unit": "tokens",
          "display": "1,050",
          "n": 24,
          "note": "Call range 176 to 3,895 tokens; not a confidence interval."
        },
        {
          "id": "retry-escalate-breakeven-cost",
          "label": "Haiku call cost, as a share of measured, at which one retry then Sonnet matches Sonnet every time (calculation)",
          "value": 0.1177,
          "unit": "ratio",
          "display": "11.8% of measured",
          "n": 8,
          "note": "All pass rates kept as observed. Arithmetic, not a prediction: a Haiku that thinks less may pass at a different rate."
        },
        {
          "id": "retry-escalate-breakeven-time",
          "label": "Haiku call time, as a share of measured, at which one retry then Sonnet matches Sonnet every time (calculation)",
          "value": 0.0543,
          "unit": "ratio",
          "display": "5.4% of measured",
          "n": 8,
          "note": "All pass rates kept as observed. Arithmetic, not a prediction."
        },
        {
          "id": "retry-escalate-sensitivity-lowest-ratio",
          "label": "Lowest cost ratio of any Haiku-first policy to Sonnet every time, across the four Haiku pass-rate settings (sensitivity range, calculation)",
          "value": 2.96,
          "unit": "ratio",
          "display": "3.0x",
          "n": 8,
          "note": "Haiku once, with Haiku’s pass rate on every task set to its Wilson upper bound."
        },
        {
          "id": "retry-escalate-joint-extreme-ratio",
          "label": "Lowest cost ratio of any Haiku-first policy to Sonnet every time, with Haiku’s pass rate at its Wilson upper bound and Sonnet’s at its Wilson lower bound on every task at once (joint extreme, calculation)",
          "value": 1.3,
          "unit": "ratio",
          "display": "1.3x",
          "n": 8,
          "note": "Haiku once. The lowest time ratio is 2.8x (Haiku once). In this setting Sonnet every time succeeds on 44% of the tasks, which no run showed (Sonnet passed 24/24). A joint extreme on 3 calls per task, not a forecast."
        },
        {
          "id": "retry-escalate-lenient-retry-cost",
          "label": "Cost per correct answer, Haiku with one retry then Sonnet, format misses counted as passes (calculation)",
          "value": 0.04591,
          "unit": "usd",
          "display": "$0.0459",
          "n": 48
        }
      ],
      "charts": [
        {
          "id": "retry-escalate-cost-per-correct",
          "title": "Expected list-price cost per correct answer, by retry policy (calculation)",
          "subtitle": "Eight hard tasks, each equally likely. A failed try is paid for. Calls recorded in Claude Code",
          "kind": "bar",
          "unit": "usd",
          "yLabel": "USD per correct answer",
          "series": [
            {
              "name": "Cost per correct answer (calculation)",
              "points": [
                {
                  "label": "Sonnet 5.5 low effort once (calculation)",
                  "value": 0.01219,
                  "n": 16,
                  "highlight": false
                },
                {
                  "label": "Sonnet 5.5 every time (calculation)",
                  "value": 0.01322,
                  "n": 24,
                  "highlight": true
                },
                {
                  "label": "Haiku, one retry, then Sonnet (calculation)",
                  "value": 0.05664,
                  "n": 48,
                  "highlight": false
                },
                {
                  "label": "Haiku once (calculation)",
                  "value": 0.06772,
                  "n": 24,
                  "highlight": false
                },
                {
                  "label": "Haiku up to 3 tries, then Sonnet (calculation)",
                  "value": 0.07193,
                  "n": 48,
                  "highlight": false
                }
              ]
            }
          ],
          "note": "Calculation, not a run. These are median-input scenarios, not measured mean costs. Total modelled list-price cost of all tries over the 8 tasks, divided by the expected number of correct answers. Per-call cost is each task’s median call at list price (reported tokens × list price); the calls ran on a flat subscription. A try runs only when the earlier tries failed the validator, and tries are independent at each task’s observed pass rate. n is the number of recorded calls behind each bar. A calculation has no confidence interval; the sensitivity chart shows a range. Highlighted: the baseline, Sonnet 5.5 every time.",
          "sourceIds": [
            "calc-repricing",
            "agent-provider-h2h-hard",
            "agent-effort-ladder",
            "price-anthropic"
          ]
        },
        {
          "id": "retry-escalate-time-per-correct",
          "title": "Expected wall time per correct answer, by retry policy (calculation)",
          "subtitle": "Tries run one after another, so a failed try adds its time",
          "kind": "bar",
          "unit": "seconds",
          "yLabel": "Seconds per correct answer",
          "series": [
            {
              "name": "Time per correct answer (calculation)",
              "points": [
                {
                  "label": "Sonnet 5.5 low effort once (calculation)",
                  "value": 7.07,
                  "n": 16,
                  "highlight": false
                },
                {
                  "label": "Sonnet 5.5 every time (calculation)",
                  "value": 8.24,
                  "n": 24,
                  "highlight": true
                },
                {
                  "label": "Haiku, one retry, then Sonnet (calculation)",
                  "value": 71.54,
                  "n": 48,
                  "highlight": false
                },
                {
                  "label": "Haiku once (calculation)",
                  "value": 91.47,
                  "n": 24,
                  "highlight": false
                },
                {
                  "label": "Haiku up to 3 tries, then Sonnet (calculation)",
                  "value": 93.42,
                  "n": 48,
                  "highlight": false
                }
              ]
            }
          ],
          "note": "Calculation, not a run. These are median-input scenarios, not measured mean times. Each task’s median total call time (CLI start-up included), summed over the tries a policy expects to run, divided by the expected number of correct answers. Tries are assumed to run one after another. Parallel tries were not tested. One host, one network. A calculation has no confidence interval. Highlighted: the baseline, Sonnet 5.5 every time.",
          "sourceIds": [
            "calc-repricing",
            "agent-provider-h2h-hard",
            "agent-effort-ladder",
            "price-anthropic"
          ]
        },
        {
          "id": "retry-escalate-success",
          "title": "Expected success rate by retry policy (calculation)",
          "subtitle": "Share of the eight hard tasks answered correctly under the strict validator",
          "kind": "bar",
          "unit": "rate",
          "polarity": "higher",
          "yLabel": "Correct answers",
          "series": [
            {
              "name": "Expected success rate",
              "points": [
                {
                  "label": "Sonnet 5.5 every time (calculation)",
                  "value": 1,
                  "n": 24,
                  "lo": 0.862,
                  "hi": 1,
                  "highlight": true
                },
                {
                  "label": "Haiku, one retry, then Sonnet (calculation)",
                  "value": 1,
                  "n": 48,
                  "highlight": false
                },
                {
                  "label": "Haiku up to 3 tries, then Sonnet (calculation)",
                  "value": 1,
                  "n": 48,
                  "highlight": false
                },
                {
                  "label": "Sonnet 5.5 low effort once (calculation)",
                  "value": 1,
                  "n": 16,
                  "lo": 0.8064,
                  "hi": 1,
                  "highlight": false
                },
                {
                  "label": "Haiku once (calculation)",
                  "value": 0.4583,
                  "n": 24,
                  "lo": 0.2789,
                  "hi": 0.6493,
                  "highlight": false
                }
              ]
            }
          ],
          "note": "Whiskers are 95% Wilson intervals on the measured strict pass rate and appear only on the three one-try policies (24/24, 11/24 and 16/16). The two escalation policies are calculations without an interval. A value of 100% means no failure was recorded: the model treats Sonnet 5.5 as always right (24/24, 95% interval 86% to 100%), so the escalation policies cannot miss. n is the number of recorded calls behind each bar.",
          "whisker": "ci95",
          "sourceIds": [
            "calc-repricing",
            "agent-provider-h2h-hard",
            "agent-effort-ladder",
            "price-anthropic"
          ]
        },
        {
          "id": "retry-escalate-haiku-by-task",
          "title": "Strict pass rate of Haiku 4.5 on each hard task",
          "subtitle": "Claude Haiku 4.5 · Claude Code, default effort. 3 calls per task",
          "kind": "bar",
          "unit": "rate",
          "polarity": "higher",
          "yLabel": "Strict passes",
          "series": [
            {
              "name": "Strict pass rate (n = 3 per task)",
              "points": [
                {
                  "label": "Interval merge fix",
                  "value": 1,
                  "lo": 0.4385,
                  "hi": 1,
                  "n": 3
                },
                {
                  "label": "DST day length",
                  "value": 0.3333,
                  "lo": 0.0615,
                  "hi": 0.7923,
                  "n": 3
                },
                {
                  "label": "CSV parser",
                  "value": 0.6667,
                  "lo": 0.2077,
                  "hi": 0.9385,
                  "n": 3
                },
                {
                  "label": "Event-loop order",
                  "value": 0,
                  "lo": 0,
                  "hi": 0.5615,
                  "n": 3
                },
                {
                  "label": "Room schedule",
                  "value": 0,
                  "lo": 0,
                  "hi": 0.5615,
                  "n": 3
                },
                {
                  "label": "SemVer regex",
                  "value": 1,
                  "lo": 0.4385,
                  "hi": 1,
                  "n": 3
                },
                {
                  "label": "Money refactor",
                  "value": 0.6667,
                  "lo": 0.2077,
                  "hi": 0.9385,
                  "n": 3
                },
                {
                  "label": "SQLite report query",
                  "value": 0,
                  "lo": 0,
                  "hi": 0.5615,
                  "n": 3
                }
              ]
            }
          ],
          "note": "Measured. Whiskers are 95% Wilson intervals on only 3 calls per task, so they are wide: 0 of 3 allows a true rate up to 56%, and 3 of 3 allows one as low as 44%. A format miss (a right answer in a code fence or with prose) is not a pass. The retry policies take each task’s rate from this chart.",
          "whisker": "ci95",
          "sourceIds": [
            "agent-provider-h2h-hard"
          ]
        },
        {
          "id": "retry-escalate-call-cost-by-task",
          "title": "List-price cost of one call, Haiku 4.5 vs Sonnet 5.5, on each hard task (calculation)",
          "subtitle": "Median of each task’s calls, default effort, Claude Code",
          "kind": "grouped-bar",
          "unit": "usd",
          "yLabel": "USD per call",
          "series": [
            {
              "name": "Claude Haiku 4.5 · Claude Code",
              "points": [
                {
                  "label": "Interval merge fix",
                  "value": 0.01789,
                  "n": 3
                },
                {
                  "label": "DST day length",
                  "value": 0.0293,
                  "n": 3
                },
                {
                  "label": "CSV parser",
                  "value": 0.02919,
                  "n": 3
                },
                {
                  "label": "Event-loop order",
                  "value": 0.03708,
                  "n": 3
                },
                {
                  "label": "Room schedule",
                  "value": 0.03578,
                  "n": 3
                },
                {
                  "label": "SemVer regex",
                  "value": 0.04217,
                  "n": 3
                },
                {
                  "label": "Money refactor",
                  "value": 0.02039,
                  "n": 3
                },
                {
                  "label": "SQLite report query",
                  "value": 0.03648,
                  "n": 3
                }
              ]
            },
            {
              "name": "Claude Sonnet 5.5 · Claude Code",
              "points": [
                {
                  "label": "Interval merge fix",
                  "value": 0.00557,
                  "n": 3
                },
                {
                  "label": "DST day length",
                  "value": 0.02532,
                  "n": 3
                },
                {
                  "label": "CSV parser",
                  "value": 0.01464,
                  "n": 3
                },
                {
                  "label": "Event-loop order",
                  "value": 0.01588,
                  "n": 3
                },
                {
                  "label": "Room schedule",
                  "value": 0.01243,
                  "n": 3
                },
                {
                  "label": "SemVer regex",
                  "value": 0.00514,
                  "n": 3
                },
                {
                  "label": "Money refactor",
                  "value": 0.00961,
                  "n": 3
                },
                {
                  "label": "SQLite report query",
                  "value": 0.01719,
                  "n": 3
                }
              ]
            }
          ],
          "note": "Calculation: reported tokens × list price for each call, then the median per task; the calls ran on a flat subscription. Haiku’s list price per token is lower. Its median calls cost more and contained more output tokens. This does not isolate the effect of effort or thinking. 3 calls per task and configuration; the table shows each call-cost range, not a confidence interval.",
          "sourceIds": [
            "calc-repricing",
            "agent-provider-h2h-hard",
            "agent-effort-ladder",
            "price-anthropic"
          ]
        },
        {
          "id": "retry-escalate-sensitivity",
          "title": "Sensitivity: cost per correct answer if Haiku’s pass rate is higher or lower (calculation)",
          "subtitle": "Dot: Haiku’s pass rate as observed. Whisker: its pass rate on every task set to its Wilson lower and upper bound at once",
          "kind": "dot-range",
          "unit": "usd",
          "yLabel": "USD per correct answer",
          "series": [
            {
              "name": "Cost per correct answer (calculation)",
              "points": [
                {
                  "label": "Sonnet 5.5 low effort once (calculation)",
                  "value": 0.01219,
                  "n": 16,
                  "highlight": false
                },
                {
                  "label": "Sonnet 5.5 every time (calculation)",
                  "value": 0.01322,
                  "n": 24,
                  "highlight": true
                },
                {
                  "label": "Haiku, one retry, then Sonnet (calculation)",
                  "value": 0.05664,
                  "lo": 0.03941,
                  "hi": 0.06807,
                  "n": 48,
                  "highlight": false
                },
                {
                  "label": "Haiku once (calculation)",
                  "value": 0.06772,
                  "lo": 0.03908,
                  "hi": 0.1834,
                  "n": 24,
                  "highlight": false
                },
                {
                  "label": "Haiku up to 3 tries, then Sonnet (calculation)",
                  "value": 0.07193,
                  "lo": 0.04149,
                  "hi": 0.09047,
                  "n": 48,
                  "highlight": false
                }
              ]
            }
          ],
          "note": "Calculation. This is a sensitivity range, not a confidence interval: each Haiku policy is recomputed with Haiku’s pass rate on all 8 tasks at once set to the lower and then the upper end of that task’s 95% Wilson interval (3 calls per task, so the ends are far apart), and the whisker spans the lowest and highest result. That is more extreme than a 95% interval on the total. The two Sonnet policies do not depend on Haiku’s pass rate and have no whisker. Highlighted: the baseline, Sonnet 5.5 every time.",
          "whisker": "minmax",
          "sourceIds": [
            "calc-repricing",
            "agent-provider-h2h-hard",
            "agent-effort-ladder",
            "price-anthropic"
          ]
        }
      ],
      "tables": [
        {
          "id": "retry-escalate-policies",
          "title": "Retry policies over the eight hard tasks (calculation)",
          "columns": [
            {
              "key": "policy",
              "label": "Policy",
              "unit": "text"
            },
            {
              "key": "tries",
              "label": "Tries in order",
              "unit": "text"
            },
            {
              "key": "success",
              "label": "Expected success",
              "unit": "rate"
            },
            {
              "key": "costPerTask",
              "label": "Expected cost per task (all tries)",
              "unit": "usd"
            },
            {
              "key": "costPerCorrect",
              "label": "Cost per correct answer",
              "unit": "usd"
            },
            {
              "key": "secPerCorrect",
              "label": "Wall time per correct answer",
              "unit": "seconds"
            },
            {
              "key": "costRatio",
              "label": "Cost vs Sonnet every time",
              "unit": "ratio"
            },
            {
              "key": "timeRatio",
              "label": "Time vs Sonnet every time",
              "unit": "ratio"
            }
          ],
          "rows": [
            {
              "policy": "Sonnet 5.5 every time (calculation)",
              "tries": "Sonnet 5.5",
              "success": 1,
              "costPerTask": 0.01322,
              "costPerCorrect": 0.01322,
              "secPerCorrect": 8.24,
              "costRatio": 1,
              "timeRatio": 1
            },
            {
              "policy": "Haiku once (calculation)",
              "tries": "Haiku 4.5",
              "success": 0.4583,
              "costPerTask": 0.03104,
              "costPerCorrect": 0.06772,
              "secPerCorrect": 91.47,
              "costRatio": 5.12,
              "timeRatio": 11.1
            },
            {
              "policy": "Haiku, one retry, then Sonnet (calculation)",
              "tries": "Haiku 4.5 → Haiku 4.5 → Sonnet 5.5",
              "success": 1,
              "costPerTask": 0.05664,
              "costPerCorrect": 0.05664,
              "secPerCorrect": 71.54,
              "costRatio": 4.28,
              "timeRatio": 8.68
            },
            {
              "policy": "Haiku up to 3 tries, then Sonnet (calculation)",
              "tries": "Haiku 4.5 → Haiku 4.5 → Haiku 4.5 → Sonnet 5.5",
              "success": 1,
              "costPerTask": 0.07193,
              "costPerCorrect": 0.07193,
              "secPerCorrect": 93.42,
              "costRatio": 5.44,
              "timeRatio": 11.33
            },
            {
              "policy": "Sonnet 5.5 low effort once (calculation)",
              "tries": "Sonnet 5.5 low",
              "success": 1,
              "costPerTask": 0.01219,
              "costPerCorrect": 0.01219,
              "secPerCorrect": 7.07,
              "costRatio": 0.92,
              "timeRatio": 0.86
            }
          ]
        },
        {
          "id": "retry-escalate-task-inputs",
          "title": "Per-task inputs: passes, 95% intervals, median call cost (calculation) and time; ranges are min–max, not confidence intervals",
          "columns": [
            {
              "key": "task",
              "label": "Task",
              "unit": "text"
            },
            {
              "key": "haikuPass",
              "label": "Claude Haiku 4.5 · Claude Code: strict passes",
              "unit": "text"
            },
            {
              "key": "haikuInterval",
              "label": "Haiku 95% interval",
              "unit": "text"
            },
            {
              "key": "haikuCost",
              "label": "Haiku median call cost",
              "unit": "usd"
            },
            {
              "key": "haikuCostRange",
              "label": "Haiku call-cost range (calculation)",
              "unit": "text"
            },
            {
              "key": "haikuTime",
              "label": "Haiku median call time",
              "unit": "seconds"
            },
            {
              "key": "haikuTimeRange",
              "label": "Haiku call-time range",
              "unit": "text"
            },
            {
              "key": "sonnetPass",
              "label": "Claude Sonnet 5.5 · Claude Code: strict passes",
              "unit": "text"
            },
            {
              "key": "sonnetInterval",
              "label": "Sonnet 95% interval",
              "unit": "text"
            },
            {
              "key": "sonnetCost",
              "label": "Sonnet median call cost",
              "unit": "usd"
            },
            {
              "key": "sonnetCostRange",
              "label": "Sonnet call-cost range (calculation)",
              "unit": "text"
            },
            {
              "key": "sonnetTime",
              "label": "Sonnet median call time",
              "unit": "seconds"
            },
            {
              "key": "sonnetTimeRange",
              "label": "Sonnet call-time range",
              "unit": "text"
            },
            {
              "key": "lowPass",
              "label": "Claude Sonnet 5.5 (low) · Claude Code: strict passes",
              "unit": "text"
            },
            {
              "key": "lowInterval",
              "label": "Sonnet (low) 95% interval",
              "unit": "text"
            },
            {
              "key": "lowCost",
              "label": "Sonnet (low) median call cost",
              "unit": "usd"
            },
            {
              "key": "lowCostRange",
              "label": "Sonnet (low) call-cost range (calculation)",
              "unit": "text"
            },
            {
              "key": "lowTime",
              "label": "Sonnet (low) median call time",
              "unit": "seconds"
            },
            {
              "key": "lowTimeRange",
              "label": "Sonnet (low) call-time range",
              "unit": "text"
            }
          ],
          "rows": [
            {
              "task": "Fix an interval-merge function (off-by-one and edge cases)",
              "haikuPass": "3/3",
              "haikuInterval": "44% to 100%",
              "haikuCost": 0.01789,
              "haikuCostRange": "$0.0148 to $0.0183",
              "haikuTime": 21.22,
              "haikuTimeRange": "18.5 s to 24.7 s",
              "sonnetPass": "3/3",
              "sonnetInterval": "44% to 100%",
              "sonnetCost": 0.00557,
              "sonnetCostRange": "$0.0056 to $0.0091",
              "sonnetTime": 2.93,
              "sonnetTimeRange": "2.4 s to 3.1 s",
              "lowPass": "2/2",
              "lowInterval": "34% to 100%",
              "lowCost": 0.00738,
              "lowCostRange": "$0.0056 to $0.0092",
              "lowTime": 2.79,
              "lowTimeRange": "2.8 s to 2.8 s"
            },
            {
              "task": "Fix a time-zone day-length function (DST)",
              "haikuPass": "1/3",
              "haikuInterval": "6% to 79%",
              "haikuCost": 0.0293,
              "haikuCostRange": "$0.0243 to $0.0308",
              "haikuTime": 39,
              "haikuTimeRange": "26.2 s to 40.6 s",
              "sonnetPass": "3/3",
              "sonnetInterval": "44% to 100%",
              "sonnetCost": 0.02532,
              "sonnetCostRange": "$0.0231 to $0.0424",
              "sonnetTime": 21.61,
              "sonnetTimeRange": "19.6 s to 34.8 s",
              "lowPass": "2/2",
              "lowInterval": "34% to 100%",
              "lowCost": 0.02395,
              "lowCostRange": "$0.0218 to $0.0261",
              "lowTime": 18.12,
              "lowTimeRange": "16.3 s to 20.0 s"
            },
            {
              "task": "Write a CSV parser (quoted newlines, strict errors)",
              "haikuPass": "2/3 (+1 format miss)",
              "haikuInterval": "21% to 94%",
              "haikuCost": 0.02919,
              "haikuCostRange": "$0.0250 to $0.0505",
              "haikuTime": 39.01,
              "haikuTimeRange": "33.8 s to 75.1 s",
              "sonnetPass": "3/3",
              "sonnetInterval": "44% to 100%",
              "sonnetCost": 0.01464,
              "sonnetCostRange": "$0.0146 to $0.0161",
              "sonnetTime": 9.56,
              "sonnetTimeRange": "8.8 s to 10.6 s",
              "lowPass": "2/2",
              "lowInterval": "34% to 100%",
              "lowCost": 0.0093,
              "lowCostRange": "$0.0091 to $0.0096",
              "lowTime": 4.36,
              "lowTimeRange": "4.3 s to 4.4 s"
            },
            {
              "task": "Predict JavaScript event-loop output order",
              "haikuPass": "0/3",
              "haikuInterval": "0% to 56%",
              "haikuCost": 0.03708,
              "haikuCostRange": "$0.0231 to $0.0395",
              "haikuTime": 54.02,
              "haikuTimeRange": "26.0 s to 56.6 s",
              "sonnetPass": "3/3",
              "sonnetInterval": "44% to 100%",
              "sonnetCost": 0.01588,
              "sonnetCostRange": "$0.0147 to $0.0161",
              "sonnetTime": 9.57,
              "sonnetTimeRange": "8.2 s to 9.9 s",
              "lowPass": "2/2",
              "lowInterval": "34% to 100%",
              "lowCost": 0.01396,
              "lowCostRange": "$0.0139 to $0.0141",
              "lowTime": 9.27,
              "lowTimeRange": "8.8 s to 9.7 s"
            },
            {
              "task": "Solve a multi-constraint room schedule",
              "haikuPass": "0/3 (+2 format misses)",
              "haikuInterval": "0% to 56%",
              "haikuCost": 0.03578,
              "haikuCostRange": "$0.0292 to $0.0444",
              "haikuTime": 54.64,
              "haikuTimeRange": "38.0 s to 68.8 s",
              "sonnetPass": "3/3",
              "sonnetInterval": "44% to 100%",
              "sonnetCost": 0.01243,
              "sonnetCostRange": "$0.0118 to $0.0126",
              "sonnetTime": 7.74,
              "sonnetTimeRange": "7.4 s to 7.8 s",
              "lowPass": "2/2",
              "lowInterval": "34% to 100%",
              "lowCost": 0.01139,
              "lowCostRange": "$0.0113 to $0.0115",
              "lowTime": 6.6,
              "lowTimeRange": "6.5 s to 6.7 s"
            },
            {
              "task": "Write a strict SemVer 2.0.0 regex",
              "haikuPass": "3/3",
              "haikuInterval": "44% to 100%",
              "haikuCost": 0.04217,
              "haikuCostRange": "$0.0373 to $0.0447",
              "haikuTime": 64.48,
              "haikuTimeRange": "54.5 s to 67.0 s",
              "sonnetPass": "3/3",
              "sonnetInterval": "44% to 100%",
              "sonnetCost": 0.00514,
              "sonnetCostRange": "$0.0051 to $0.0079",
              "sonnetTime": 2.49,
              "sonnetTimeRange": "2.3 s to 3.7 s",
              "lowPass": "2/2",
              "lowInterval": "34% to 100%",
              "lowCost": 0.00514,
              "lowCostRange": "$0.0051 to $0.0051",
              "lowTime": 3.18,
              "lowTimeRange": "2.9 s to 3.5 s"
            },
            {
              "task": "Refactor to remove duplication, keep 20 tests green",
              "haikuPass": "2/3",
              "haikuInterval": "21% to 94%",
              "haikuCost": 0.02039,
              "haikuCostRange": "$0.0179 to $0.0216",
              "haikuTime": 15.91,
              "haikuTimeRange": "15.3 s to 17.2 s",
              "sonnetPass": "3/3",
              "sonnetInterval": "44% to 100%",
              "sonnetCost": 0.00961,
              "sonnetCostRange": "$0.0094 to $0.0151",
              "sonnetTime": 3.57,
              "sonnetTimeRange": "3.4 s to 7.2 s",
              "lowPass": "2/2",
              "lowInterval": "34% to 100%",
              "lowCost": 0.00945,
              "lowCostRange": "$0.0094 to $0.0095",
              "lowTime": 4.44,
              "lowTimeRange": "3.7 s to 5.2 s"
            },
            {
              "task": "Write a SQLite reporting query (fan-out, ties, boundaries)",
              "haikuPass": "0/3 (+2 format misses)",
              "haikuInterval": "0% to 56%",
              "haikuCost": 0.03648,
              "haikuCostRange": "$0.0289 to $0.0406",
              "haikuTime": 47.11,
              "haikuTimeRange": "38.9 s to 56.1 s",
              "sonnetPass": "3/3",
              "sonnetInterval": "44% to 100%",
              "sonnetCost": 0.01719,
              "sonnetCostRange": "$0.0157 to $0.0192",
              "sonnetTime": 8.47,
              "sonnetTimeRange": "7.6 s to 8.9 s",
              "lowPass": "2/2",
              "lowInterval": "34% to 100%",
              "lowCost": 0.01695,
              "lowCostRange": "$0.0167 to $0.0172",
              "lowTime": 7.8,
              "lowTimeRange": "7.7 s to 7.9 s"
            }
          ]
        },
        {
          "id": "retry-escalate-sensitivity-table",
          "title": "Sensitivity of each policy to Haiku’s pass rate (calculation)",
          "columns": [
            {
              "key": "policy",
              "label": "Policy",
              "unit": "text"
            },
            {
              "key": "setting",
              "label": "Haiku pass rate set to",
              "unit": "text"
            },
            {
              "key": "success",
              "label": "Expected success",
              "unit": "rate"
            },
            {
              "key": "costPerCorrect",
              "label": "Cost per correct answer",
              "unit": "usd"
            },
            {
              "key": "secPerCorrect",
              "label": "Wall time per correct answer",
              "unit": "seconds"
            }
          ],
          "rows": [
            {
              "policy": "Sonnet 5.5 every time (calculation)",
              "setting": "Does not depend on Haiku",
              "success": 1,
              "costPerCorrect": 0.01322,
              "secPerCorrect": 8.24
            },
            {
              "policy": "Haiku once (calculation)",
              "setting": "Haiku pass rate at the Wilson lower bound (every task)",
              "success": 0.1692,
              "costPerCorrect": 0.1834,
              "secPerCorrect": 247.74
            },
            {
              "policy": "Haiku once (calculation)",
              "setting": "Haiku pass rate as observed",
              "success": 0.4583,
              "costPerCorrect": 0.06772,
              "secPerCorrect": 91.47
            },
            {
              "policy": "Haiku once (calculation)",
              "setting": "Haiku pass rate at the Wilson upper bound (every task)",
              "success": 0.7942,
              "costPerCorrect": 0.03908,
              "secPerCorrect": 52.79
            },
            {
              "policy": "Haiku once (calculation)",
              "setting": "Haiku format misses counted as passes (lenient reading)",
              "success": 0.6667,
              "costPerCorrect": 0.04655,
              "secPerCorrect": 62.89
            },
            {
              "policy": "Haiku, one retry, then Sonnet (calculation)",
              "setting": "Haiku pass rate at the Wilson lower bound (every task)",
              "success": 1,
              "costPerCorrect": 0.06807,
              "secPerCorrect": 84.27
            },
            {
              "policy": "Haiku, one retry, then Sonnet (calculation)",
              "setting": "Haiku pass rate as observed",
              "success": 1,
              "costPerCorrect": 0.05664,
              "secPerCorrect": 71.54
            },
            {
              "policy": "Haiku, one retry, then Sonnet (calculation)",
              "setting": "Haiku pass rate at the Wilson upper bound (every task)",
              "success": 1,
              "costPerCorrect": 0.03941,
              "secPerCorrect": 52.64
            },
            {
              "policy": "Haiku, one retry, then Sonnet (calculation)",
              "setting": "Haiku format misses counted as passes (lenient reading)",
              "success": 1,
              "costPerCorrect": 0.04591,
              "secPerCorrect": 59.5
            },
            {
              "policy": "Haiku up to 3 tries, then Sonnet (calculation)",
              "setting": "Haiku pass rate at the Wilson lower bound (every task)",
              "success": 1,
              "costPerCorrect": 0.09047,
              "secPerCorrect": 115.27
            },
            {
              "policy": "Haiku up to 3 tries, then Sonnet (calculation)",
              "setting": "Haiku pass rate as observed",
              "success": 1,
              "costPerCorrect": 0.07193,
              "secPerCorrect": 93.42
            },
            {
              "policy": "Haiku up to 3 tries, then Sonnet (calculation)",
              "setting": "Haiku pass rate at the Wilson upper bound (every task)",
              "success": 1,
              "costPerCorrect": 0.04149,
              "secPerCorrect": 56.17
            },
            {
              "policy": "Haiku up to 3 tries, then Sonnet (calculation)",
              "setting": "Haiku format misses counted as passes (lenient reading)",
              "success": 1,
              "costPerCorrect": 0.05263,
              "secPerCorrect": 69.47
            },
            {
              "policy": "Sonnet 5.5 low effort once (calculation)",
              "setting": "Does not depend on Haiku",
              "success": 1,
              "costPerCorrect": 0.01219,
              "secPerCorrect": 7.07
            }
          ]
        }
      ],
      "related": [
        "hard-model-head-to-head",
        "effort-ladder"
      ]
    },
    {
      "slug": "harder-tasks-head-to-head",
      "title": "GPT-6.1 Sol vs Claude Opus 5.5, Sonnet 5.5 and Haiku 4.5 on 4 harder tasks",
      "seoTitle": "GPT-6.1 Sol vs Claude Opus 5.5 on harder tasks",
      "description": "56 counted calls on 4 harder tasks with strict validators: GPT-6.1 Sol, Claude Opus 5.5, Sonnet 5.5, Haiku 4.5. Pass rate, intervals, speed, cost.",
      "question": "On a task set built so that Claude Sonnet 5.5 did not pass it every time, does pass rate separate GPT-6.1 Sol (Codex CLI) from Claude Opus 5.5, Sonnet 5.5 and Haiku 4.5 (Claude Code)?",
      "answer": "22 of 56 counted calls passed strictly (39%, 95% Wilson interval 28% to 52%) on 4 tasks. The Sonnet pilot did not pass these twice. GPT-6.1 Sol (medium) passed 11/16 strictly (69%, 95% interval 44% to 86%). Opus 5.5 passed 5/12 strictly (42%, 95% interval 19% to 68%). Sonnet 5.5 passed 6/16 strictly (38%, 95% interval 18% to 61%). Haiku 4.5 passed 0/12 strictly (0%, 95% interval 0% to 24%). GPT-6.1 Sol (medium) is ahead of Haiku 4.5 (the 95% intervals do not overlap). The other 5 of 6 pairs overlap, so this set cannot rank them. The lenient reading counts format misses. Opus 5.5 is also ahead of Haiku 4.5 on the lenient reading. Opus 5.5 6/12 (n = 12, 95% Wilson interval 25.4% to 74.6%); Haiku 4.5 0/12 (n = 12, 95% Wilson interval 0.0% to 24.2%). The intervals miss by 1.1 points (calculation), so this is fragile. Haiku 4.5 passed none (a floor for that configuration on this set). No configuration passed every call across the full set. Some per-task cells still hit a ceiling; see the task chart. Of 34 non-passes, 1 was a format miss with the right answer. Another 21 were wrong answers. 12 gave no answer; 10 hit the 300 s timeout. 11 of 56 calls tried a tool although tools were off. Per configuration: GPT-6.1 Sol (medium) 0 of 16, Opus 5.5 5 of 12, Sonnet 5.5 5 of 16 and Haiku 4.5 1 of 12. None of them passed. The Codex runner also tells the model not to call tools; the Claude Code runner does not. Median total time per completed call (wrong answers and format misses included; timeouts and tool-call parse errors excluded; ranges are not intervals): GPT-6.1 Sol (medium) 120.2 s (n = 13, range 46.2 s to 273.5 s). Opus 5.5 80.3 s (n = 9, range 3.8 s to 279.5 s). Sonnet 5.5 70.4 s (n = 12, range 4.3 s to 210.1 s). Haiku 4.5 109.0 s (n = 10, range 25.7 s to 223.9 s). Every configuration’s fastest-to-slowest range overlaps every other, so the medians describe this run and are not a tested ranking. Cost is a list-price calculation; the calls ran on subscriptions. The lowest recorded lower bound was GPT-6.1 Sol (medium) · Codex CLI: $0.083 (a lower bound: 3 timed-out calls report no tokens; $0.100 if each had cost a median call, an assumption). Selection effect: the study picked tasks with mixed or failed Sonnet pilot results. Sonnet passed 2 of 8 pilot calls on the kept tasks and 6 of 16 counted calls. The counted calls are new calls. Selection can produce this pattern; the run does not establish its cause.",
      "date": "2026-10-07",
      "updated": "2026-10-07",
      "tags": [
        "head-to-head",
        "harder-tasks",
        "gpt-6-1-sol",
        "codex-cli",
        "claude-sonnet",
        "claude-opus",
        "claude-haiku",
        "reasoning",
        "selection-effect",
        "format-misses",
        "latency"
      ],
      "method": [
        "The original protocol file predates the first probe and counted call. This review checked file birth times. Later changes are amendments. This follows the hard head-to-head (/benchmarks/hard-model-head-to-head), which hit a pass-rate ceiling for the strong models.",
        "The study started with 16 candidate tasks, written in two rounds (12 first, then 4 replacements; code, SQL, reasoning, spec, simulation, numeric). Each is one call with no tools, an exact output format and a deterministic validator that runs in a sandbox without network. Claude Sonnet 5.5 at default effort ran each candidate twice (the pilot, 32 calls). It passed 12 of the 16 candidates 2 of 2. The study dropped them.",
        "The dropped candidates include 10 of 10 code, SQL, spec, numeric and simulation tasks.",
        "The protocol declared the selection rule before the pilot. Keep at most 8 tasks: first those Sonnet passed 1 of 2, then 0 of 2. Drop tasks with 2 of 2 passes. Only 1 of the first 12 qualified, below the threshold of 6. The study wrote 4 replacement candidates once and piloted them twice; 3 qualified.",
        "The counted set is 4 tasks: 10x10 nonogram; Sudoku, 22 givens; 6x6 Skyscrapers; Seeded shuffle output. The study ran no second round of replacements.",
        "Controls ran in stages. The first 12 candidates had controls before the first probe. After a module-export validator defect, controls ran again and stored pilot replies were re-scored without new calls. Final controls covered all 16 candidates before the replacement pilot and all counted calls.\n\n16/16 references pass. 66/66 wrong answers fail. 16/16 wrapped references are format misses. Independent solvers checked unique reasoning answers. Python integers confirmed the code-reading answer.",
        "Counted cells: GPT-6.1 Sol (medium) · Codex CLI (16 calls, 4 per task). Claude Opus 5.5 · Claude Code (12 calls, 3 per task). Claude Sonnet 5.5 · Claude Code (16 calls, 4 per task). Claude Haiku 4.5 · Claude Code (12 calls, 3 per task). Rep-major, round-robin order, one call at a time per account, 300 s timeout.",
        "Strict pass: the whole reply, trimmed, passes the validator. Every prompt states the output format and says no other text. Format miss: the strict check fails, but an extracted answer passes the same validator. The extractor reads fenced blocks, answer lines or grid rows. For exact tasks it reads only the last answer-shaped candidate, not an earlier guess. Reported apart from wrong answers, never as a pass.",
        "An error or timeout counts as a non-pass. Tool use is a separate flag, not a score. It marks tool-call markup or a CLI tool-call parse error.",
        "Isolation: fresh empty working folder, tools off, no MCP servers, no session persistence, one turn. The Codex runner adds a developer instruction to answer directly and not to call tools. The Claude Code runner adds no such instruction. Claude Code ran with an output-token cap setting of 16,000; the Codex CLI had none.",
        "Default effort means the effort flag was not passed. GPT-6.1 Sol ran at medium effort.",
        "Counted calls: Claude Code: 40 counted calls, 40 reached a model; Codex CLI: 16 counted calls, 16 reached a model. The study trimmed no calls. The study repeated no counted key. No run reported a usage or rate limit. The Claude CLI reported its own failed retry on tool-call parse errors; those receipts remain failures.\n\nThe Claude batch stopped 1 time on a tool-call parse error. A later amendment kept that failure but allowed the lane to continue. The lane repeated no counted key.",
        "Uncounted probes: 2, kept apart from 32 pilot calls and 56 counted attempts. Call caps, including probes and pilots: Claude Code: 73/105 calls; Codex CLI: 17/33 calls.",
        "Cost per strict pass is a calculation: total reported-token cost divided by strict passes. It includes failed calls with tokens and prices cache reads and writes. Timeouts report no tokens, so the cost is a lower bound."
      ],
      "caveats": [
        "Selection effect: the study picked tasks that Sonnet did not pass twice in the pilot. Sonnet passed 2 of 8 pilot calls on the kept tasks and 6 of 16 counted calls (calculation: 25% and 38%; the intervals overlap). Selection can produce this pattern, but the run does not establish its cause.",
        "Opus and Haiku share Sonnet’s family and CLI, so tasks picked against Sonnet may be hard for them for the same reasons. GPT-6.1 Sol was not selected against, which can widen the gap between Sol and the Claude rows. A fixed set picked before any run would be fairer.",
        "Small samples: GPT-6.1 Sol (medium) n = 16, Opus 5.5 n = 12, Sonnet 5.5 n = 16, Haiku 4.5 n = 12. Per-task cells have 3 to 4 calls. The intervals are wide, and the calls on one task are not independent, so the intervals are likely narrower than the truth.",
        "Each row pairs a CLI with a model on one subscription. The CLIs use different subscriptions, system prompts and start-up steps. A Claude-vs-GPT row compares route + model pairs, not the models alone.",
        "Both CLIs disabled tools. Every prompt asks for the answer only. The runners differ in one way: the Codex runner adds a developer instruction not to call tools, and the Claude Code runner adds none. 11 of 40 Claude Code calls tried a tool anyway. By model: Opus 5.5 5 of 12, Sonnet 5.5 5 of 16 and Haiku 4.5 1 of 12.\n\nNone of them passed. None of the 16 Codex CLI calls did. We did not test the instruction, so we cannot say how much of the gap it explains. With tools on, the Claude Code models might have run code and passed on these tasks. This study does not test that either.",
        "The runner stopped 10 of 56 calls at the 300 s timeout and counted them as no answer. Counts: GPT-6.1 Sol (medium) 3, Opus 5.5 1, Sonnet 5.5 4 and Haiku 4.5 2. A longer limit might turn some into answers, passes or fails.\n\nTiming medians cover completed calls only. They include wrong answers and format misses. They omit only timeouts and tool-call parse errors in this run.",
        "Claude Code ran with an output-token cap setting of 16,000. Reported totals exceeded 16,000 in 11 of 40 Claude Code calls. The setting did not bound reported totals. We cannot tell whether it cut any reply. The tool-call parse errors came at 16,746 and 16,859 reported output tokens. The Codex CLI calls had no such cap.",
        "The kept tasks are mostly long exact-search or computation tasks. In the pilot Claude Sonnet 5.5 passed 10 of 10 code, SQL, spec, numeric and simulation candidates 2 of 2. This set says little about everyday coding.",
        "Claude Haiku 4.5 · Claude Code passed none: a floor on this set, not a general rating.",
        "Strict format rules decide part of the result: a reply with extra text fails. The chart shows the lenient reading next to the strict score.",
        "Arena servers shared the Mac during part of the run. Host load was not controlled. CLI timings include start-up and each CLI’s system prompt. These times do not isolate model speed.",
        "The protocol first said to drop tasks with validator defects. An amendment instead repaired the JavaScript validators and re-scored stored pilot replies. None of those code-writing tasks entered the counted set.",
        "No configuration passed every call across the full set. Some per-task cells hit a ceiling: Sonnet and Opus on the nonogram, and Sol on Skyscrapers. Their perfect cells do not prove equal ability.",
        "Wilson intervals treat calls as independent. The same four tasks repeat, so these are descriptive call-level intervals, not population intervals for coding tasks. The study ran no task-level paired significance test.",
        "Default effort means the effort flag was not passed; the CLI chose. Median reasoning tokens per completed call, where the CLI reports them (n and ranges are in the token chart): GPT-6.1 Sol (medium) 4,971, Opus 5.5 8,352, Sonnet 5.5 6,557, Haiku 4.5 12,483. They are part of the output tokens.",
        "List-price costs are calculations; the calls used flat subscriptions. Opus 5.5 figures are provisional: its cache-read price is under re-check. Timeout calls report no tokens and are not priced. Unpriced calls: GPT-6.1 Sol (medium) 3 of 16, Opus 5.5 1 of 12 and Sonnet 5.5 4 of 16. Those cells show a lower bound on cost per pass."
      ],
      "sourceIds": [
        "agent-harder-tasks",
        "calc-repricing",
        "price-anthropic",
        "price-openai"
      ],
      "stats": [
        {
          "id": "harder-h2h-pass-all",
          "label": "Counted calls that passed strictly (harder set)",
          "value": 0.3929,
          "unit": "rate",
          "display": "39% (22/56)",
          "n": 56,
          "ci": [
            0.2758,
            0.5237
          ]
        },
        {
          "id": "harder-h2h-correct-all",
          "label": "Counted calls with a correct answer, format misses included (lenient reading)",
          "value": 0.4107,
          "unit": "rate",
          "display": "41% (23/56)",
          "n": 56,
          "ci": [
            0.2917,
            0.5412
          ]
        },
        {
          "id": "harder-h2h-pass-sol",
          "label": "GPT-6.1 Sol (medium) (Codex CLI): strict pass rate on the harder tasks",
          "value": 0.6875,
          "unit": "rate",
          "display": "69% (11/16)",
          "n": 16,
          "ci": [
            0.444,
            0.8584
          ]
        },
        {
          "id": "harder-h2h-pass-opus",
          "label": "Opus 5.5 (Claude Code): strict pass rate on the harder tasks",
          "value": 0.4167,
          "unit": "rate",
          "display": "42% (5/12)",
          "n": 12,
          "ci": [
            0.1933,
            0.6805
          ]
        },
        {
          "id": "harder-h2h-pass-sonnet",
          "label": "Sonnet 5.5 (Claude Code): strict pass rate on the harder tasks",
          "value": 0.375,
          "unit": "rate",
          "display": "38% (6/16)",
          "n": 16,
          "ci": [
            0.1848,
            0.6136
          ]
        },
        {
          "id": "harder-h2h-pass-haiku",
          "label": "Haiku 4.5 (Claude Code): strict pass rate on the harder tasks",
          "value": 0,
          "unit": "rate",
          "display": "0% (0/12)",
          "n": 12,
          "ci": [
            0,
            0.2425
          ]
        },
        {
          "id": "harder-h2h-format-misses",
          "label": "Non-passes that were format misses, not wrong answers",
          "value": 1,
          "unit": "count",
          "display": "1 of 34 non-passes (21 wrong answers, 12 no answer)",
          "n": 34
        },
        {
          "id": "harder-h2h-tool-attempts",
          "label": "Counted calls that tried a tool although tools were off (a behaviour, not a quality score)",
          "value": 0.1964,
          "unit": "rate",
          "display": "20% (11/56)",
          "n": 56,
          "ci": [
            0.1134,
            0.3184
          ],
          "note": "Per configuration: GPT-6.1 Sol (medium) 0 of 16, Opus 5.5 5 of 12, Sonnet 5.5 5 of 16 and Haiku 4.5 1 of 12. None of these calls passed. The flag comes from the reply text: tool-call markup, or a CLI message that a tool call could not be parsed. The Codex runner also tells the model not to call tools; the Claude Code runner does not."
        },
        {
          "id": "harder-h2h-timeouts",
          "label": "Counted calls that ran past the 300 s limit and gave no answer",
          "value": 0.1786,
          "unit": "rate",
          "display": "18% (10/56)",
          "n": 56,
          "ci": [
            0.1,
            0.2984
          ],
          "note": "Per configuration: GPT-6.1 Sol (medium) 3 of 16, Opus 5.5 1 of 12, Sonnet 5.5 4 of 16 and Haiku 4.5 2 of 12. A timeout counts as a non-pass."
        },
        {
          "id": "harder-h2h-pilot-sonnet",
          "label": "Pilot (not scored): strict passes of Claude Sonnet 5.5 on all candidate tasks",
          "value": 0.8125,
          "unit": "rate",
          "display": "81% (26/32)",
          "n": 32,
          "ci": [
            0.6469,
            0.9111
          ],
          "note": "Two calls per candidate. Its rates have a selection effect and repeated-task dependence. The pilot decided which tasks were kept; it is not part of any counted cell."
        },
        {
          "id": "harder-h2h-pilot-sonnet-kept",
          "label": "Pilot (not scored): strict passes of Claude Sonnet 5.5 on the tasks that were kept",
          "value": 0.25,
          "unit": "rate",
          "display": "25% (2/8)",
          "n": 8,
          "ci": [
            0.0715,
            0.5907
          ],
          "note": "The same four tasks the counted calls use. The counted Sonnet rate is in harder-h2h-pass-sonnet. Selection against Sonnet predicts a lower pilot rate than a fresh rate."
        },
        {
          "id": "harder-h2h-tasks-kept",
          "label": "Candidate tasks kept for the counted set",
          "value": 4,
          "unit": "count",
          "display": "4 of 16 (Sonnet passed 12 candidates 2 of 2, including 10 of 10 code, SQL, spec, numeric and simulation tasks)",
          "n": 16
        },
        {
          "id": "harder-h2h-cheapest-per-pass",
          "label": "Lowest recorded cost lower bound per strict pass (calculation)",
          "value": 0.08293,
          "unit": "usd",
          "display": "GPT-6.1 Sol (medium) · Codex CLI: $0.0829 (lower bound)",
          "n": 16,
          "note": "All passing configurations have unpriced timeout calls. Unknown costs can change their order. This is a recorded lower bound, not a cost ranking."
        }
      ],
      "charts": [
        {
          "id": "harder-h2h-pass-rate",
          "title": "Pass rate on 4 harder tasks",
          "subtitle": "Strict: the reply passes as given. Lenient: a correct answer in the wrong format also counts",
          "kind": "dot-range",
          "unit": "rate",
          "polarity": "higher",
          "yLabel": "Passed",
          "whisker": "ci95",
          "series": [
            {
              "name": "Strict pass",
              "points": [
                {
                  "label": "GPT-6.1 Sol (medium) · Codex CLI",
                  "value": 0.6875,
                  "lo": 0.444,
                  "hi": 0.8584,
                  "n": 16
                },
                {
                  "label": "Claude Opus 5.5 · Claude Code",
                  "value": 0.4167,
                  "lo": 0.1933,
                  "hi": 0.6805,
                  "n": 12
                },
                {
                  "label": "Claude Sonnet 5.5 · Claude Code",
                  "value": 0.375,
                  "lo": 0.1848,
                  "hi": 0.6136,
                  "n": 16
                },
                {
                  "label": "Claude Haiku 4.5 · Claude Code",
                  "value": 0,
                  "lo": 0,
                  "hi": 0.2425,
                  "n": 12
                }
              ]
            },
            {
              "name": "Lenient (format misses counted)",
              "points": [
                {
                  "label": "GPT-6.1 Sol (medium) · Codex CLI",
                  "value": 0.6875,
                  "lo": 0.444,
                  "hi": 0.8584,
                  "n": 16
                },
                {
                  "label": "Claude Opus 5.5 · Claude Code",
                  "value": 0.5,
                  "lo": 0.2538,
                  "hi": 0.7462,
                  "n": 12
                },
                {
                  "label": "Claude Sonnet 5.5 · Claude Code",
                  "value": 0.375,
                  "lo": 0.1848,
                  "hi": 0.6136,
                  "n": 16
                },
                {
                  "label": "Claude Haiku 4.5 · Claude Code",
                  "value": 0,
                  "lo": 0,
                  "hi": 0.2425,
                  "n": 12
                }
              ]
            }
          ],
          "note": "Whiskers are 95% Wilson intervals. Strict passes decide the result. The lenient reading shows which failures contain a correct answer in the wrong format. A format miss never counts as a pass. The study picked tasks that Sonnet did not pass twice in a pilot. This selection can lower a Sonnet row.\n\nCounted calls are new calls.",
          "sourceIds": [
            "agent-harder-tasks"
          ]
        },
        {
          "id": "harder-h2h-outcomes",
          "title": "What happened on every call",
          "subtitle": "Counts per configuration: strict passes, format misses, wrong answers and calls with no answer",
          "kind": "stacked-bar",
          "unit": "count",
          "yLabel": "Calls",
          "series": [
            {
              "name": "Strict pass",
              "points": [
                {
                  "label": "GPT-6.1 Sol (medium) · Codex CLI",
                  "value": 11,
                  "n": 16
                },
                {
                  "label": "Claude Opus 5.5 · Claude Code",
                  "value": 5,
                  "n": 12
                },
                {
                  "label": "Claude Sonnet 5.5 · Claude Code",
                  "value": 6,
                  "n": 16
                },
                {
                  "label": "Claude Haiku 4.5 · Claude Code",
                  "value": 0,
                  "n": 12
                }
              ]
            },
            {
              "name": "Format miss (correct answer, wrong format)",
              "points": [
                {
                  "label": "GPT-6.1 Sol (medium) · Codex CLI",
                  "value": 0,
                  "n": 16
                },
                {
                  "label": "Claude Opus 5.5 · Claude Code",
                  "value": 1,
                  "n": 12
                },
                {
                  "label": "Claude Sonnet 5.5 · Claude Code",
                  "value": 0,
                  "n": 16
                },
                {
                  "label": "Claude Haiku 4.5 · Claude Code",
                  "value": 0,
                  "n": 12
                }
              ]
            },
            {
              "name": "Wrong answer",
              "points": [
                {
                  "label": "GPT-6.1 Sol (medium) · Codex CLI",
                  "value": 2,
                  "n": 16
                },
                {
                  "label": "Claude Opus 5.5 · Claude Code",
                  "value": 3,
                  "n": 12
                },
                {
                  "label": "Claude Sonnet 5.5 · Claude Code",
                  "value": 6,
                  "n": 16
                },
                {
                  "label": "Claude Haiku 4.5 · Claude Code",
                  "value": 10,
                  "n": 12
                }
              ]
            },
            {
              "name": "No answer (timeout or error)",
              "points": [
                {
                  "label": "GPT-6.1 Sol (medium) · Codex CLI",
                  "value": 3,
                  "n": 16
                },
                {
                  "label": "Claude Opus 5.5 · Claude Code",
                  "value": 3,
                  "n": 12
                },
                {
                  "label": "Claude Sonnet 5.5 · Claude Code",
                  "value": 4,
                  "n": 16
                },
                {
                  "label": "Claude Haiku 4.5 · Claude Code",
                  "value": 2,
                  "n": 12
                }
              ]
            }
          ],
          "note": "A format miss fails strictly but has an extracted answer that passes the same validator. Extra working or grids can cause this outcome. It is not a pass. A call with no answer is a timeout or an error; it counts as a non-pass.",
          "sourceIds": [
            "agent-harder-tasks"
          ]
        },
        {
          "id": "harder-h2h-tool-attempts",
          "title": "Calls that tried a tool although tools were off",
          "subtitle": "Share of calls whose reply held tool-call markup or whose tool call the CLI could not parse",
          "kind": "dot-range",
          "unit": "rate",
          "yLabel": "Calls with a tool attempt",
          "polarity": "none",
          "whisker": "ci95",
          "series": [
            {
              "name": "Tool attempt",
              "points": [
                {
                  "label": "GPT-6.1 Sol (medium) · Codex CLI",
                  "value": 0,
                  "lo": 0,
                  "hi": 0.1936,
                  "n": 16
                },
                {
                  "label": "Claude Opus 5.5 · Claude Code",
                  "value": 0.4167,
                  "lo": 0.1933,
                  "hi": 0.6805,
                  "n": 12
                },
                {
                  "label": "Claude Sonnet 5.5 · Claude Code",
                  "value": 0.3125,
                  "lo": 0.1416,
                  "hi": 0.556,
                  "n": 16
                },
                {
                  "label": "Claude Haiku 4.5 · Claude Code",
                  "value": 0.0833,
                  "lo": 0.0149,
                  "hi": 0.3539,
                  "n": 12
                }
              ]
            }
          ],
          "note": "Every prompt asked for the answer only. Both CLIs disabled tools. The Codex runner adds a developer instruction not to call tools. The Claude Code runner adds no such instruction. The validator decides the score. Tool-call markup is a separate flag; a flagged call can still pass if its whole reply meets the validator.\n\nIt is a behaviour, not a quality score: more is not better. No call with a tool attempt passed. Whiskers are 95% Wilson intervals. The flag comes from the reply text, which is not published.",
          "sourceIds": [
            "agent-harder-tasks"
          ]
        },
        {
          "id": "harder-h2h-pass-by-task",
          "title": "Strict pass rate by task",
          "subtitle": "One bar per configuration and task; each bar rests on only a few calls, so the 95% intervals are wide",
          "kind": "grouped-bar",
          "unit": "rate",
          "polarity": "higher",
          "yLabel": "Strict pass rate",
          "whisker": "ci95",
          "series": [
            {
              "name": "GPT-6.1 Sol (medium) · Codex CLI",
              "points": [
                {
                  "label": "10x10 nonogram",
                  "value": 0.75,
                  "lo": 0.3006,
                  "hi": 0.9544,
                  "n": 4
                },
                {
                  "label": "Sudoku, 22 givens",
                  "value": 0.25,
                  "lo": 0.0456,
                  "hi": 0.6994,
                  "n": 4
                },
                {
                  "label": "6x6 Skyscrapers",
                  "value": 1,
                  "lo": 0.5101,
                  "hi": 1,
                  "n": 4
                },
                {
                  "label": "Seeded shuffle output",
                  "value": 0.75,
                  "lo": 0.3006,
                  "hi": 0.9544,
                  "n": 4
                }
              ]
            },
            {
              "name": "Claude Opus 5.5 · Claude Code",
              "points": [
                {
                  "label": "10x10 nonogram",
                  "value": 1,
                  "lo": 0.4385,
                  "hi": 1,
                  "n": 3
                },
                {
                  "label": "Sudoku, 22 givens",
                  "value": 0,
                  "lo": 0,
                  "hi": 0.5615,
                  "n": 3
                },
                {
                  "label": "6x6 Skyscrapers",
                  "value": 0.3333,
                  "lo": 0.0615,
                  "hi": 0.7923,
                  "n": 3
                },
                {
                  "label": "Seeded shuffle output",
                  "value": 0.3333,
                  "lo": 0.0615,
                  "hi": 0.7923,
                  "n": 3
                }
              ]
            },
            {
              "name": "Claude Sonnet 5.5 · Claude Code",
              "points": [
                {
                  "label": "10x10 nonogram",
                  "value": 1,
                  "lo": 0.5101,
                  "hi": 1,
                  "n": 4
                },
                {
                  "label": "Sudoku, 22 givens",
                  "value": 0,
                  "lo": 0,
                  "hi": 0.4899,
                  "n": 4
                },
                {
                  "label": "6x6 Skyscrapers",
                  "value": 0,
                  "lo": 0,
                  "hi": 0.4899,
                  "n": 4
                },
                {
                  "label": "Seeded shuffle output",
                  "value": 0.5,
                  "lo": 0.15,
                  "hi": 0.85,
                  "n": 4
                }
              ]
            },
            {
              "name": "Claude Haiku 4.5 · Claude Code",
              "points": [
                {
                  "label": "10x10 nonogram",
                  "value": 0,
                  "lo": 0,
                  "hi": 0.5615,
                  "n": 3
                },
                {
                  "label": "Sudoku, 22 givens",
                  "value": 0,
                  "lo": 0,
                  "hi": 0.5615,
                  "n": 3
                },
                {
                  "label": "6x6 Skyscrapers",
                  "value": 0,
                  "lo": 0,
                  "hi": 0.5615,
                  "n": 3
                },
                {
                  "label": "Seeded shuffle output",
                  "value": 0,
                  "lo": 0,
                  "hi": 0.5615,
                  "n": 3
                }
              ]
            }
          ],
          "note": "Each bar is 3 to 4 calls, so one call moves a bar by a quarter or a third. Whiskers are 95% Wilson intervals; with this few calls they overlap almost everywhere.",
          "sourceIds": [
            "agent-harder-tasks"
          ]
        },
        {
          "id": "harder-h2h-total-latency",
          "title": "Total time per call on harder tasks",
          "subtitle": "Median per configuration; whiskers = fastest and slowest call",
          "kind": "dot-range",
          "unit": "seconds",
          "yLabel": "Seconds",
          "whisker": "minmax",
          "series": [
            {
              "name": "Total time per call",
              "points": [
                {
                  "label": "GPT-6.1 Sol (medium) · Codex CLI",
                  "value": 120.24,
                  "lo": 46.24,
                  "hi": 273.46,
                  "n": 13,
                  "highlight": false
                },
                {
                  "label": "Claude Opus 5.5 · Claude Code",
                  "value": 80.34,
                  "lo": 3.82,
                  "hi": 279.5,
                  "n": 9,
                  "highlight": false
                },
                {
                  "label": "Claude Sonnet 5.5 · Claude Code",
                  "value": 70.43,
                  "lo": 4.32,
                  "hi": 210.08,
                  "n": 12,
                  "highlight": false
                },
                {
                  "label": "Claude Haiku 4.5 · Claude Code",
                  "value": 108.98,
                  "lo": 25.73,
                  "hi": 223.95,
                  "n": 10,
                  "highlight": false
                }
              ]
            }
          ],
          "note": "Median and range over the calls that completed. Completed calls include wrong answers and format misses. Only timeouts and tool-call parse errors are excluded from this run’s timings. Both count as non-passes in the outcomes chart. One Mac, one network, one session.\n\nArena servers shared the Mac during part of the run. Host load was not controlled, so these times cannot isolate model speed. Whiskers are a range, not a confidence interval. Times include the CLI start-up and the CLI’s own system prompt. Highlighted: configurations that passed every call.",
          "sourceIds": [
            "agent-harder-tasks"
          ]
        },
        {
          "id": "harder-h2h-output-tokens",
          "title": "Output tokens per call on harder tasks",
          "subtitle": "Median per configuration; reasoning tokens as the CLI reports them",
          "kind": "grouped-bar",
          "unit": "tokens",
          "yLabel": "Tokens",
          "polarity": "none",
          "whisker": "minmax",
          "series": [
            {
              "name": "Output tokens",
              "points": [
                {
                  "label": "GPT-6.1 Sol (medium) · Codex CLI",
                  "value": 4994,
                  "lo": 2099,
                  "hi": 13413,
                  "n": 13
                },
                {
                  "label": "Claude Opus 5.5 · Claude Code",
                  "value": 8420,
                  "lo": 279,
                  "hi": 40044,
                  "n": 9
                },
                {
                  "label": "Claude Sonnet 5.5 · Claude Code",
                  "value": 9287,
                  "lo": 407,
                  "hi": 27921,
                  "n": 12
                },
                {
                  "label": "Claude Haiku 4.5 · Claude Code",
                  "value": 12508,
                  "lo": 2965,
                  "hi": 26532,
                  "n": 10
                }
              ]
            },
            {
              "name": "Reasoning tokens",
              "points": [
                {
                  "label": "GPT-6.1 Sol (medium) · Codex CLI",
                  "value": 4971,
                  "lo": 2070,
                  "hi": 13372,
                  "n": 13
                },
                {
                  "label": "Claude Opus 5.5 · Claude Code",
                  "value": 8352,
                  "lo": 21,
                  "hi": 9897,
                  "n": 9
                },
                {
                  "label": "Claude Sonnet 5.5 · Claude Code",
                  "value": 6557,
                  "lo": 63,
                  "hi": 27902,
                  "n": 12
                },
                {
                  "label": "Claude Haiku 4.5 · Claude Code",
                  "value": 12483,
                  "lo": 2924,
                  "hi": 26510,
                  "n": 10
                }
              ]
            }
          ],
          "note": "Medians and minimum-to-maximum token ranges cover completed calls only. Ranges are not confidence intervals. The chart omits unknown reasoning counts. Reasoning tokens are part of the output tokens where the CLI reports them. Their content is never captured.\n\nClaude Code used an output-token cap setting of 16,000. Some reported totals exceeded it.\n\nCodex CLI had no cap. More tokens is not better or worse by itself.",
          "sourceIds": [
            "agent-harder-tasks"
          ]
        },
        {
          "id": "harder-h2h-cost-per-pass",
          "title": "List-price cost per strict pass on harder tasks (calculation)",
          "subtitle": "All calls in a configuration, failures and format misses included, divided by its strict passes",
          "kind": "bar",
          "unit": "usd",
          "yLabel": "USD per strict pass",
          "series": [
            {
              "name": "Cost per strict pass",
              "points": [
                {
                  "label": "GPT-6.1 Sol (medium) · Codex CLI",
                  "value": 0.08293,
                  "n": 16,
                  "highlight": true
                },
                {
                  "label": "Claude Sonnet 5.5 · Claude Code",
                  "value": 0.23843,
                  "n": 16,
                  "highlight": false
                },
                {
                  "label": "Claude Opus 5.5 · Claude Code",
                  "value": 0.59333,
                  "n": 12,
                  "highlight": false
                }
              ]
            }
          ],
          "note": "Calculation, not a bill: reported tokens × list price; the calls ran on flat subscriptions. A failed call still costs, so a lower pass rate raises the cost per pass. Timeout calls report no tokens and are not priced. Unpriced calls: GPT-6.1 Sol (medium) 3 of 16, Opus 5.5 1 of 12 and Sonnet 5.5 4 of 16. These cells show a lower bound.\n\nAssume each unpriced call cost its cell’s median priced call. This sensitivity calculation gives GPT-6.1 Sol (medium) $0.100, Opus 5.5 $0.633 and Sonnet 5.5 $0.303. Opus 5.5 figures are provisional: its cache-read price is under re-check.\n\nHighlights mark the observed frontier of these lower-bound costs. Unknown timeout costs can change it; this is not a cost ranking. Claude Haiku 4.5 · Claude Code had no strict pass, so it has no cost per pass.",
          "sourceIds": [
            "agent-harder-tasks",
            "calc-repricing",
            "price-anthropic",
            "price-openai"
          ]
        },
        {
          "id": "harder-h2h-frontier",
          "title": "Observed quality vs cost frontier (calculation)",
          "subtitle": "Strict pass rate against list-price cost per strict pass",
          "kind": "scatter",
          "unit": "rate",
          "polarity": "higher",
          "xLabel": "USD per strict pass (list-price calculation)",
          "yLabel": "Strict pass rate",
          "whisker": "ci95",
          "series": [
            {
              "name": "Codex CLI",
              "points": [
                {
                  "label": "GPT-6.1 Sol (medium) · Codex CLI",
                  "value": 0.6875,
                  "lo": 0.444,
                  "hi": 0.8584,
                  "n": 16,
                  "x": 0.08293,
                  "highlight": true
                }
              ]
            },
            {
              "name": "Claude Code",
              "points": [
                {
                  "label": "Claude Opus 5.5 · Claude Code",
                  "value": 0.4167,
                  "lo": 0.1933,
                  "hi": 0.6805,
                  "n": 12,
                  "x": 0.59333,
                  "highlight": false
                },
                {
                  "label": "Claude Sonnet 5.5 · Claude Code",
                  "value": 0.375,
                  "lo": 0.1848,
                  "hi": 0.6136,
                  "n": 16,
                  "x": 0.23843,
                  "highlight": false
                }
              ]
            }
          ],
          "note": "Upper-left has a higher observed pass rate and lower recorded cost per pass. Highlights mark the observed frontier of lower-bound costs. Unknown timeout costs can change it. This is not a tested ranking. Frontier: GPT-6.1 Sol (medium) · Codex CLI.\n\nCosts are calculations from tokens; calls that timed out are not priced (GPT-6.1 Sol (medium) 3 of 16, Opus 5.5 1 of 12 and Sonnet 5.5 4 of 16), so those cells are lower bounds. The 95% Wilson intervals are listed below; the plot shows point estimates.\n\nStrict rates: GPT-6.1 Sol (medium) 11/16 (n = 16, 95% interval 44.4% to 85.8%); Opus 5.5 5/12 (n = 12, 95% interval 19.3% to 68.0%); Sonnet 5.5 6/16 (n = 16, 95% interval 18.5% to 61.4%).",
          "sourceIds": [
            "agent-harder-tasks",
            "calc-repricing",
            "price-anthropic",
            "price-openai"
          ]
        }
      ],
      "tables": [
        {
          "id": "harder-h2h-cells",
          "title": "Cells: configuration, calls and outcomes",
          "columns": [
            {
              "key": "config",
              "label": "Configuration",
              "unit": "text"
            },
            {
              "key": "calls",
              "label": "Calls",
              "unit": "count"
            },
            {
              "key": "strict",
              "label": "Strict passes",
              "unit": "text"
            },
            {
              "key": "interval",
              "label": "95% Wilson interval",
              "unit": "text"
            },
            {
              "key": "formatMisses",
              "label": "Format misses",
              "unit": "count"
            },
            {
              "key": "wrong",
              "label": "Wrong answers",
              "unit": "count"
            },
            {
              "key": "noAnswer",
              "label": "No answer",
              "unit": "count"
            },
            {
              "key": "priced",
              "label": "Calls with a token report",
              "unit": "count"
            },
            {
              "key": "completed",
              "label": "Completed calls (timing and token n)",
              "unit": "count"
            },
            {
              "key": "medianTotal",
              "label": "Median completed-call total (s)",
              "unit": "seconds"
            },
            {
              "key": "rangeTotal",
              "label": "Completed-call time range (s, not an interval)",
              "unit": "text"
            },
            {
              "key": "p95Total",
              "label": "p95 total (s)",
              "unit": "seconds"
            },
            {
              "key": "medianOut",
              "label": "Median output tokens",
              "unit": "tokens"
            }
          ],
          "rows": [
            {
              "config": "GPT-6.1 Sol (medium) · Codex CLI",
              "calls": 16,
              "strict": "11/16",
              "interval": "44% to 86%",
              "formatMisses": 0,
              "wrong": 2,
              "noAnswer": 3,
              "priced": 13,
              "completed": 13,
              "rangeTotal": "46.24 to 273.46",
              "medianTotal": 120.24,
              "p95Total": 240.94,
              "medianOut": 4994
            },
            {
              "config": "Claude Opus 5.5 · Claude Code",
              "calls": 12,
              "strict": "5/12",
              "interval": "19% to 68%",
              "formatMisses": 1,
              "wrong": 3,
              "noAnswer": 3,
              "priced": 11,
              "completed": 9,
              "rangeTotal": "3.82 to 279.5",
              "medianTotal": 80.34,
              "p95Total": 208.56,
              "medianOut": 8420
            },
            {
              "config": "Claude Sonnet 5.5 · Claude Code",
              "calls": 16,
              "strict": "6/16",
              "interval": "18% to 61%",
              "formatMisses": 0,
              "wrong": 6,
              "noAnswer": 4,
              "priced": 12,
              "completed": 12,
              "rangeTotal": "4.32 to 210.08",
              "medianTotal": 70.43,
              "p95Total": 208.72,
              "medianOut": 9287
            },
            {
              "config": "Claude Haiku 4.5 · Claude Code",
              "calls": 12,
              "strict": "0/12",
              "interval": "0% to 24%",
              "formatMisses": 0,
              "wrong": 10,
              "noAnswer": 2,
              "priced": 10,
              "completed": 10,
              "rangeTotal": "25.73 to 223.95",
              "medianTotal": 108.98,
              "p95Total": 202.31,
              "medianOut": 12508
            }
          ]
        },
        {
          "id": "harder-h2h-tasks",
          "title": "Candidate tasks, pilot result and selection (prompts not shown)",
          "columns": [
            {
              "key": "task",
              "label": "Task",
              "unit": "text"
            },
            {
              "key": "kind",
              "label": "Kind",
              "unit": "text"
            },
            {
              "key": "checks",
              "label": "Validator checks",
              "unit": "count"
            },
            {
              "key": "round",
              "label": "Pilot round",
              "unit": "text"
            },
            {
              "key": "pilot",
              "label": "Sonnet 5.5 pilot (strict passes)",
              "unit": "text"
            },
            {
              "key": "selected",
              "label": "In the counted set",
              "unit": "text"
            },
            {
              "key": "controls",
              "label": "Controls (wrong answers rejected, wrapped reference flagged)",
              "unit": "text"
            }
          ],
          "rows": [
            {
              "task": "Write a TTL and LRU cache class (random operation sequences)",
              "kind": "code",
              "checks": 13,
              "round": "first 12 candidates",
              "pilot": "2/2",
              "selected": "no",
              "controls": "5/5; yes"
            },
            {
              "task": "Write a glob matcher (braces, **, character sets, hidden files)",
              "kind": "code",
              "checks": 68,
              "round": "first 12 candidates",
              "pilot": "2/2",
              "selected": "no",
              "controls": "4/4; yes"
            },
            {
              "task": "Write a unified diff (minimal script, fixed tie-break, hunk headers)",
              "kind": "code",
              "checks": 20,
              "round": "first 12 candidates",
              "pilot": "2/2",
              "selected": "no",
              "controls": "4/4; yes"
            },
            {
              "task": "Evaluate Python-style integer expressions (precedence, chained comparisons)",
              "kind": "code",
              "checks": 64,
              "round": "first 12 candidates",
              "pilot": "2/2",
              "selected": "no",
              "controls": "5/5; yes"
            },
            {
              "task": "SQLite sessions report (gaps and islands, median, logout rule)",
              "kind": "SQL",
              "checks": 7,
              "round": "first 12 candidates",
              "pilot": "2/2",
              "selected": "no",
              "controls": "5/5; yes"
            },
            {
              "task": "SQLite as-of price and currency report (missing days, rounding)",
              "kind": "SQL",
              "checks": 14,
              "round": "first 12 candidates",
              "pilot": "2/2",
              "selected": "no",
              "controls": "5/5; yes"
            },
            {
              "task": "Solve a 6x6 Skyscrapers puzzle (14 clues, one solution)",
              "kind": "reasoning",
              "checks": 1,
              "round": "first 12 candidates",
              "pilot": "0/2 (+2 format misses)",
              "selected": "yes",
              "controls": "3/3; yes"
            },
            {
              "task": "Pick the most profitable jobs for two machines (one optimum)",
              "kind": "reasoning",
              "checks": 1,
              "round": "first 12 candidates",
              "pilot": "2/2",
              "selected": "no",
              "controls": "3/3; yes"
            },
            {
              "task": "Decode UTF-8 with one U+FFFD per maximal subpart",
              "kind": "spec",
              "checks": 35,
              "round": "first 12 candidates",
              "pilot": "2/2",
              "selected": "no",
              "controls": "4/4; yes"
            },
            {
              "task": "Next run of a cron expression in UTC (day-of-month or day-of-week rule)",
              "kind": "spec",
              "checks": 37,
              "round": "first 12 candidates",
              "pilot": "2/2",
              "selected": "no",
              "controls": "4/4; yes"
            },
            {
              "task": "Simulate a retry queue (priorities, timeouts, backoff, ties)",
              "kind": "simulation",
              "checks": 1,
              "round": "first 12 candidates",
              "pilot": "2/2",
              "selected": "no",
              "controls": "4/4; yes"
            },
            {
              "task": "Correctly rounded sum of doubles (ties, overflow, negative zero)",
              "kind": "numeric",
              "checks": 32,
              "round": "first 12 candidates",
              "pilot": "2/2",
              "selected": "no",
              "controls": "4/4; yes"
            },
            {
              "task": "Solve a 10x10 nonogram (one solution)",
              "kind": "reasoning",
              "checks": 1,
              "round": "replacement candidates",
              "pilot": "1/2",
              "selected": "yes",
              "controls": "4/4; yes"
            },
            {
              "task": "Pick the most profitable jobs for three machines (30 jobs, one optimum)",
              "kind": "reasoning",
              "checks": 1,
              "round": "replacement candidates",
              "pilot": "2/2",
              "selected": "no",
              "controls": "5/5; yes"
            },
            {
              "task": "Predict the output of a seeded shuffle (32-bit integer arithmetic)",
              "kind": "code reading",
              "checks": 1,
              "round": "replacement candidates",
              "pilot": "0/2",
              "selected": "yes",
              "controls": "4/4; yes"
            },
            {
              "task": "Solve a 9x9 Sudoku with 22 givens (one solution)",
              "kind": "reasoning",
              "checks": 1,
              "round": "replacement candidates",
              "pilot": "1/2",
              "selected": "yes",
              "controls": "3/3; yes"
            }
          ]
        }
      ],
      "related": [
        "hard-model-head-to-head",
        "effort-ladder"
      ]
    },
    {
      "slug": "voice-agent-latency-budget",
      "title": "Voice agent latency budget: component calculations, not a measured turn",
      "seoTitle": "Voice agent latency budget: calculated component times",
      "description": "Calculation: component times against assumed 300 ms, 800 ms and 1,500 ms budgets. No voice turn or audio was measured.",
      "question": "Which decision steps fit inside a voice agent’s turn budget, using only measured times?",
      "answer": "Rules and Jev 1.13 fit all three assumed budgets at the median and p95. LLM routers through the Claude Code CLI fit none at the median. Calculation: the budgets are 300 ms, 800 ms and 1,500 ms per step. At the slow end, 3, 3 and 4 of 14 measured steps fit these budgets, respectively. At the median, 3, 3 and 8 fit. The deterministic policy took a median 1.42 µs (p95 2.33 µs, n = 20,000). Its maximum was 2.5 ms; its minimum was not retained. Jev 1.13 took a median 136.5 ms over HTTPS (p95 195.7 ms, n = 246). Its range was 100.9 ms to 297.3 ms. Calculation: this uses 46% of a 300 ms budget at the median and 65% at p95. Sonnet 5.5 routing through Claude Code took a median 2,598 ms (p95 4,298 ms, n = 82). Its range was 1,993 ms to 5,583 ms. Its median exceeds a 1,500 ms budget. GPT-6 Luna had the lowest observed API median for first output: 823 ms (n = 5). Its range was 506 ms to 1,371 ms. Calculation: its median is over an 800 ms budget. Its slowest run is over that budget. API ranges overlap, so this is not a speed ranking. These ranges are not confidence intervals. All times come from other studies on one Mac, over different routes. The source runs did not measure a voice turn or audio.",
      "date": "2026-10-06",
      "updated": "2026-10-06",
      "tags": [
        "voice-agent",
        "latency",
        "latency-budget",
        "routing",
        "jev",
        "thought-experiment",
        "calculation"
      ],
      "method": [
        "Thought experiment on recorded times. This study makes no new model calls. Every time comes from a run that another study published.",
        "Budgets: 300 ms, 800 ms, 1,500 ms per decision step (assumption: a budget a voice team might set). Nothing in the data says what a product needs; change the budget and the table gives the share and the count for it.",
        "Deterministic routing policy: the policy timing of the routing overhead run (latencyUs p50 and p95, converted to milliseconds). Rule-based System One decision: the recorded decision times of that run (millisecond resolution, record write excluded).",
        "Jev 1.13: this study checks the per-call rows against the live-run summary. The rows record client wall time over HTTPS. They give n, median and p95 (latencyMs calls). The first call uses a fresh connection (coldFirstCall). This calculation subtracts the later-call median (callsWithoutColdFirst) from the first-call time. The run starts at run.startedAt and ends at run.endedAt. The client uses one Apple Silicon Mac on a home network.",
        "Claude Sonnet 5.5 and Claude Haiku 4.5 as routers: median and interpolated p95 from the routing extract (latencyMs wallP50 and wallP95). The overhead extract supplies n and the observed limits, through the Claude Code CLI, one call at a time. CLI start-up: first model output on a one-word prompt, from the routing overhead run (firstModelEventMs; 5 runs per CLI).",
        "Model first output: this study uses the matched provider explorer cohort for the fixed exact reply. Each configuration has 5 runs with firstUsefulMs timings. Routes are the OpenAI API and the Codex CLI. This study also selects the Claude Code configuration with the lowest observed first-output median in the 5-task head-to-head. Its timing sample includes completed validator failures. Failed or untimed calls have no successful first-output time.",
        "Slow end: the 95th percentile where a step has 30 or more runs, and the slowest run where it has fewer. The slowest run is stricter than a 95th percentile.",
        "Calculations: share of a budget = time ÷ budget, at the median and at the slow end. A step fits when its time is at or below the budget. Runs that fit back to back = the budget ÷ the time, rounded down, counted up to 1,000. A router in front of a model adds the two medians and the two slow ends, as a scenario. Neither sum is a measured percentile or a guaranteed upper bound for the pair.",
        "The charts separate steps with a median under 1,000 ms from the rest so each axis stays readable. Separate charts show median-to-p95 bands and observed min-to-max ranges. A range maximum is not a p95."
      ],
      "caveats": [
        "Timing samples omit 0 failed or untimed matched explorer calls and 0 failed or untimed Claude head-to-head calls. A failure is not a fast successful step. No pass rate is estimated here.",
        "The overhead extract uses nearest-rank quantiles. For LLM routers this study instead uses the conventional median and interpolated p95 from the routing extract. Per-iteration policy times and individual rule-record times are not retained, so those summaries cannot be independently rebuilt.",
        "The rule-arm latency is captured before the decision record is written. It excludes that write, despite the source summary label.",
        "The policy timing excludes database reads, policy merge and decision-record insertion. Its minimum was not retained. The rule-record timing also lacks a retained minimum.",
        "The Claude head-to-head tasks hit a pass-rate ceiling. This study selects the lowest observed Claude median after the run; ranges overlap and this selection is not a speed ranking. The accounts could run concurrently on the shared Mac.",
        "The 30-run cutoff is a display rule. It does not make a sample p95 a reliable production tail estimate. Some recorded calls exceed the p95; the table keeps their maxima visible.",
        "The budgets are assumptions. A voice product may need less or more; this study does not say what listeners accept.",
        "The steps differ in route and in what they time. Routers are whole decisions; models are the time to the first useful text, as the harness records it. First output is not necessarily enough text to start speech. A router through a CLI is a deployment choice, not a model limit.",
        "All times were recorded on one Mac, not on a server near the vendor, and on different days. Vendor latency changes over a day, and a server near the API would see different times.",
        "Small samples: 5 runs per model configuration, 5 per CLI start-up, 15 for the Claude Code configuration. Their slow end is the slowest run, a range and not a confidence interval.",
        "The imported explorer evidence has no frozen pre-run protocol. Those receipts do not establish that its controls and call caps were set before the run.",
        "The explorer uses a one-line answer; CLI start-up uses a one-word prompt. The selected Claude Code cell uses five short tasks, including code. Longer prompts and replies can change first-output time. No speech or audio was measured.",
        "No speech recognition, speech synthesis, turn detection or network round trip to a user is timed here. API wall times include the client-to-provider network. The steps are only the decision and the first model output.",
        "Calculations add medians and add slow ends across runs that were not made together. Medians do not add up exactly, and sums of p95 values do not guarantee a pair p95 or an upper bound. Repeated-step counts assume the same duration each time; they do not give a probability of fitting.",
        "Jev was timed in one run of 35 seconds, from one client machine on one network path, with calls one at a time. A hosted API can be faster or slower at another hour or from another region. The Jev decision cases were revised against Jev’s own answers, so its accuracy has a home advantage; this study uses only its time, not its accuracy."
      ],
      "sourceIds": [
        "calc-latency-budget",
        "agent-routing-overhead",
        "agent-routing",
        "agent-provider-explorer",
        "agent-provider-h2h",
        "agent-jev-live"
      ],
      "stats": [
        {
          "id": "latency-budget-steps-counted",
          "label": "Decision and first-output steps compared",
          "value": 14,
          "unit": "count",
          "display": "14 steps",
          "n": 14,
          "note": "5 under 1,000 ms at the median, 9 at 1,000 ms or more. Every time is a measured median from another study."
        },
        {
          "id": "latency-budget-fit-300-slow",
          "label": "Steps that fit a 300 ms budget at the slow end (calculation)",
          "value": 3,
          "unit": "count",
          "display": "3 of 14 (median: 3 of 14)",
          "n": 14,
          "note": "Budget (assumption: a budget a voice team might set). Slow end: the 95th percentile for steps with 30 or more runs, the slowest run for the rest."
        },
        {
          "id": "latency-budget-fit-800-slow",
          "label": "Steps that fit an 800 ms budget at the slow end (calculation)",
          "value": 3,
          "unit": "count",
          "display": "3 of 14 (median: 3 of 14)",
          "n": 14,
          "note": "Budget (assumption: a budget a voice team might set). Slow end: the 95th percentile for steps with 30 or more runs, the slowest run for the rest."
        },
        {
          "id": "latency-budget-fit-1500-slow",
          "label": "Steps that fit a 1,500 ms budget at the slow end (calculation)",
          "value": 4,
          "unit": "count",
          "display": "4 of 14 (median: 8 of 14)",
          "n": 14,
          "note": "Budget (assumption: a budget a voice team might set). Slow end: the 95th percentile for steps with 30 or more runs, the slowest run for the rest."
        },
        {
          "id": "latency-budget-jev-share-300",
          "label": "Share of a 300 ms budget that one Jev 1.13 decision uses at the median (calculation)",
          "value": 45.5,
          "unit": "percent",
          "display": "46% at the median, 65% at p95",
          "n": 246,
          "note": "Median 136.5 ms, p95 195.7 ms over HTTPS from one Mac on a home network. Budget (assumption: a budget a voice team might set)."
        },
        {
          "id": "latency-budget-jev-in-sequence-800",
          "label": "Back-to-back Jev 1.13 decisions that fit an 800 ms budget at the slow end (calculation)",
          "value": 4,
          "unit": "count",
          "display": "4 (5 at the median)",
          "n": 246,
          "note": "Budget (assumption: a budget a voice team might set)."
        },
        {
          "id": "latency-budget-sonnet-router-share-1500",
          "label": "Share of a 1,500 ms budget that one Sonnet 5.5 router decision uses at the median (calculation)",
          "value": 173,
          "unit": "percent",
          "display": "173% at the median, 287% at p95",
          "n": 82,
          "note": "Median 2,598 ms, p95 4,298 ms through the Claude Code CLI. Budget (assumption: a budget a voice team might set)."
        },
        {
          "id": "latency-budget-fastest-model-first-output",
          "label": "Lowest observed API median time to the first output of a one-line answer",
          "value": 823,
          "unit": "ms",
          "display": "823 ms (slowest run 1,371 ms)",
          "n": 5,
          "note": "GPT-6 Luna through the OpenAI API. Range 506 ms to 1,371 ms, n = 5; not a confidence interval. API ranges overlap, so this is not a ranking."
        },
        {
          "id": "latency-budget-codex-cli-added",
          "label": "Difference in median first-output time across 3 Codex model setups: Codex CLI minus API (calculation)",
          "value": 1964,
          "unit": "ms",
          "display": "+1,964 to +2,880 ms",
          "n": 5,
          "note": "Median through the Codex CLI minus median through the OpenAI API, for the same model, effort and one-line prompt (GPT-6 Luna: +1,964 ms; GPT-6.1 Sol (effort low): +2,880 ms; GPT-6.1 Sol (effort high): +2,445 ms). For each setup the slowest API run is below the CLI median. A calculation across runs of 5 per setup; the samples are small."
        },
        {
          "id": "latency-budget-jev-model-left-1500",
          "label": "Time left of a 1,500 ms budget after Jev 1.13 and the API cell with the lowest observed median, at the median (calculation)",
          "value": 540.5,
          "unit": "ms",
          "display": "540.5 ms left (sum of medians 959.5 ms)",
          "n": 5,
          "note": "Separate samples: Jev n = 246; API n = 5. Jev median 136.5 ms plus GPT-6 Luna median 823 ms through the OpenAI API. At the slow ends the sum is 1,566.7 ms, 66.7 ms over. This arithmetic remainder is not measured audio latency. API timings already include the client-to-provider network. Budget (assumption: a budget a voice team might set). A calculation across runs that were not made together."
        },
        {
          "id": "latency-budget-jev-cold-call",
          "label": "First Jev call minus the later-call median (calculation)",
          "value": 88.3,
          "unit": "ms",
          "display": "88.3 ms once",
          "n": 1,
          "note": "The first call used a fresh connection and took 224.7 ms; the later-call median was 136.4 ms (n = 245). One first call, so no range. This difference does not isolate connection cost or server warmth."
        }
      ],
      "charts": [
        {
          "id": "latency-budget-fast-steps",
          "title": "Steps that take under a second, against three assumed budgets (median to p95)",
          "subtitle": "Median per step; whiskers = median to p95; budget dots are assumptions",
          "kind": "dot-range",
          "unit": "ms",
          "yLabel": "Time per step",
          "whisker": "p50-p95",
          "series": [
            {
              "name": "Time per step",
              "points": [
                {
                  "label": "Deterministic routing policy (Agent, in process)",
                  "value": 0.00142,
                  "lo": 0.00142,
                  "hi": 0.00233,
                  "n": 20000
                },
                {
                  "label": "Rule-based System One decision (record write excluded)",
                  "value": 1,
                  "lo": 1,
                  "hi": 2,
                  "n": 419
                },
                {
                  "label": "Jev 1.13 (TypeSafe, direct HTTPS)",
                  "value": 136.5,
                  "lo": 136.5,
                  "hi": 195.7,
                  "n": 246
                }
              ]
            },
            {
              "name": "Budget (assumption: a budget a voice team might set)",
              "points": [
                {
                  "label": "Budget 300 ms",
                  "value": 300
                },
                {
                  "label": "Budget 800 ms",
                  "value": 800
                },
                {
                  "label": "Budget 1,500 ms",
                  "value": 1500
                }
              ]
            }
          ],
          "note": "These 5 steps have a median under 1,000 ms. Decisions are whole calls; model steps are the time to the first useful output of a one-line answer. The dot is the median. The whisker runs from the median to the 95th percentile (30 or more runs per step). Observed limits stay in the measured-step table; some minima were not retained. A whisker is not a confidence interval. The budget dots are assumptions (assumption: a budget a voice team might set); they are not measured.",
          "sourceIds": [
            "agent-routing-overhead",
            "agent-routing",
            "agent-provider-explorer",
            "agent-provider-h2h",
            "agent-jev-live"
          ]
        },
        {
          "id": "latency-budget-fast-steps-ranges",
          "title": "Steps that take under a second, against three assumed budgets (observed ranges)",
          "subtitle": "Median per step; whiskers = fastest to slowest run; budget dots are assumptions",
          "kind": "dot-range",
          "unit": "ms",
          "yLabel": "Time per step",
          "whisker": "minmax",
          "series": [
            {
              "name": "Time per step",
              "points": [
                {
                  "label": "OpenAI API · GPT-6 Luna · none (first output, one-line answer)",
                  "value": 823,
                  "lo": 506,
                  "hi": 1371,
                  "n": 5
                },
                {
                  "label": "OpenAI API · GPT-6.1 Sol · low (first output, one-line answer)",
                  "value": 873,
                  "lo": 835,
                  "hi": 1742,
                  "n": 5
                }
              ]
            },
            {
              "name": "Budget (assumption: a budget a voice team might set)",
              "points": [
                {
                  "label": "Budget 300 ms",
                  "value": 300
                },
                {
                  "label": "Budget 800 ms",
                  "value": 800
                },
                {
                  "label": "Budget 1,500 ms",
                  "value": 1500
                }
              ]
            }
          ],
          "note": "These 5 steps have a median under 1,000 ms. Decisions are whole calls; model steps are the time to the first useful output of a one-line answer. The dot is the median. The whisker runs from the fastest to the slowest observed run (fewer than 30 runs per step). The maximum is not a p95 estimate. A whisker is not a confidence interval. The budget dots are assumptions (assumption: a budget a voice team might set); they are not measured.",
          "sourceIds": [
            "agent-routing-overhead",
            "agent-routing",
            "agent-provider-explorer",
            "agent-provider-h2h",
            "agent-jev-live"
          ]
        },
        {
          "id": "latency-budget-slow-steps",
          "title": "Steps that take a second or more, against three assumed budgets (median to p95)",
          "subtitle": "Median per step; whiskers = median to p95; budget dots are assumptions",
          "kind": "dot-range",
          "unit": "ms",
          "yLabel": "Time per step",
          "whisker": "p50-p95",
          "series": [
            {
              "name": "Time per step",
              "points": [
                {
                  "label": "Claude Sonnet 5.5 (router, effort low, via Claude Code)",
                  "value": 2598,
                  "lo": 2598,
                  "hi": 4298,
                  "n": 82
                },
                {
                  "label": "Claude Haiku 4.5 (router, thinking on, via Claude Code)",
                  "value": 12674,
                  "lo": 12674,
                  "hi": 34413,
                  "n": 82
                }
              ]
            },
            {
              "name": "Budget (assumption: a budget a voice team might set)",
              "points": [
                {
                  "label": "Budget 300 ms",
                  "value": 300
                },
                {
                  "label": "Budget 800 ms",
                  "value": 800
                },
                {
                  "label": "Budget 1,500 ms",
                  "value": 1500
                }
              ]
            }
          ],
          "note": "These 9 steps have a median of 1,000 ms or more. Routers are whole calls through the Claude Code CLI; the rest are the time to the first output of a model through a CLI or the API. The dot is the median. The whisker runs from the median to the 95th percentile (30 or more runs per step). Observed limits stay in the measured-step table; some minima were not retained. A whisker is not a confidence interval. The budget dots are assumptions (assumption: a budget a voice team might set); they are not measured.",
          "sourceIds": [
            "agent-routing-overhead",
            "agent-routing",
            "agent-provider-explorer",
            "agent-provider-h2h",
            "agent-jev-live"
          ]
        },
        {
          "id": "latency-budget-slow-steps-ranges",
          "title": "Steps that take a second or more, against three assumed budgets (observed ranges)",
          "subtitle": "Median per step; whiskers = fastest to slowest run; budget dots are assumptions",
          "kind": "dot-range",
          "unit": "ms",
          "yLabel": "Time per step",
          "whisker": "minmax",
          "series": [
            {
              "name": "Time per step",
              "points": [
                {
                  "label": "Claude Code · Claude Fable 5.1 (first output, 5 short tasks)",
                  "value": 1196,
                  "lo": 947,
                  "hi": 7902,
                  "n": 15
                },
                {
                  "label": "OpenAI API · GPT-6.1 Sol · high (first output, one-line answer)",
                  "value": 1341,
                  "lo": 1259,
                  "hi": 2115,
                  "n": 5
                },
                {
                  "label": "Claude Code · Claude Haiku 4.5 (start-up, first model output)",
                  "value": 1461,
                  "lo": 1206,
                  "hi": 2308,
                  "n": 5
                },
                {
                  "label": "Codex CLI · GPT-6 Luna · none (first output, one-line answer)",
                  "value": 2787,
                  "lo": 2461,
                  "hi": 3418,
                  "n": 5
                },
                {
                  "label": "Codex CLI · GPT-6.1 Sol · low (first output, one-line answer)",
                  "value": 3753,
                  "lo": 3435,
                  "hi": 4103,
                  "n": 5
                },
                {
                  "label": "Codex CLI · GPT-6.1 Sol · high (first output, one-line answer)",
                  "value": 3786,
                  "lo": 3366,
                  "hi": 4296,
                  "n": 5
                },
                {
                  "label": "Codex CLI (start-up, first model output)",
                  "value": 5059,
                  "lo": 4391,
                  "hi": 5478,
                  "n": 5
                }
              ]
            },
            {
              "name": "Budget (assumption: a budget a voice team might set)",
              "points": [
                {
                  "label": "Budget 300 ms",
                  "value": 300
                },
                {
                  "label": "Budget 800 ms",
                  "value": 800
                },
                {
                  "label": "Budget 1,500 ms",
                  "value": 1500
                }
              ]
            }
          ],
          "note": "These 9 steps have a median of 1,000 ms or more. Routers are whole calls through the Claude Code CLI; the rest are the time to the first output of a model through a CLI or the API. The dot is the median. The whisker runs from the fastest to the slowest observed run (fewer than 30 runs per step). The maximum is not a p95 estimate. A whisker is not a confidence interval. The budget dots are assumptions (assumption: a budget a voice team might set); they are not measured.",
          "sourceIds": [
            "agent-routing-overhead",
            "agent-routing",
            "agent-provider-explorer",
            "agent-provider-h2h",
            "agent-jev-live"
          ]
        },
        {
          "id": "latency-budget-fit",
          "title": "How many of the steps fit each assumed budget (calculation)",
          "subtitle": "Count of 14 measured steps; n = steps",
          "kind": "grouped-bar",
          "unit": "count",
          "yLabel": "Steps that fit (of 14)",
          "series": [
            {
              "name": "Fits at the median",
              "points": [
                {
                  "label": "300 ms budget",
                  "value": 3,
                  "n": 14
                },
                {
                  "label": "800 ms budget",
                  "value": 3,
                  "n": 14
                },
                {
                  "label": "1,500 ms budget",
                  "value": 8,
                  "n": 14
                }
              ]
            },
            {
              "name": "Fits at the slow end",
              "points": [
                {
                  "label": "300 ms budget",
                  "value": 3,
                  "n": 14
                },
                {
                  "label": "800 ms budget",
                  "value": 3,
                  "n": 14
                },
                {
                  "label": "1,500 ms budget",
                  "value": 4,
                  "n": 14
                }
              ]
            }
          ],
          "note": "A calculation on measured times, with budgets of 300 ms, 800 ms and 1,500 ms (assumption: a budget a voice team might set). A step fits when its median, or its slow end, is at or below the budget. The slow end is the 95th percentile for steps with 30 or more runs and the slowest run for the rest, which is stricter. The steps differ in route and in what they time, so read the count as a summary of the table below, not as a ranking.",
          "sourceIds": [
            "calc-latency-budget",
            "agent-routing-overhead",
            "agent-routing",
            "agent-provider-explorer",
            "agent-provider-h2h",
            "agent-jev-live"
          ]
        }
      ],
      "tables": [
        {
          "id": "latency-budget-measured-steps",
          "title": "Every step and the time it was measured to take",
          "columns": [
            {
              "key": "step",
              "label": "Step",
              "unit": "text"
            },
            {
              "key": "covers",
              "label": "What the time covers",
              "unit": "text"
            },
            {
              "key": "route",
              "label": "Route",
              "unit": "text"
            },
            {
              "key": "n",
              "label": "Runs",
              "unit": "count"
            },
            {
              "key": "p50",
              "label": "Median",
              "unit": "ms"
            },
            {
              "key": "range",
              "label": "Observed range (not an interval)",
              "unit": "text"
            },
            {
              "key": "slow",
              "label": "Slow end",
              "unit": "ms"
            },
            {
              "key": "slowKind",
              "label": "Slow end is",
              "unit": "text"
            },
            {
              "key": "origin",
              "label": "Where the time comes from",
              "unit": "text"
            }
          ],
          "rows": [
            {
              "step": "Deterministic routing policy (Agent, in process)",
              "covers": "pure in-process policy decision; no database work",
              "route": "in process, no model call",
              "n": 20000,
              "p50": 0.00142,
              "range": "Minimum not retained to 2.5 ms (n = 20,000)",
              "slow": 0.00233,
              "slowKind": "95th percentile",
              "origin": "Routing overhead run: in-process policy timing, p50 and p95 (µs)"
            },
            {
              "step": "Rule-based System One decision (record write excluded)",
              "covers": "whole decision, request to answer",
              "route": "in process, record write excluded",
              "n": 419,
              "p50": 1,
              "range": "Minimum not retained to 3 ms (n = 419)",
              "slow": 2,
              "slowKind": "95th percentile",
              "origin": "Routing overhead run: recorded System One decisions (millisecond resolution)"
            },
            {
              "step": "Jev 1.13 (TypeSafe, direct HTTPS)",
              "covers": "whole decision, request to answer",
              "route": "direct HTTPS, home network",
              "n": 246,
              "p50": 136.5,
              "range": "100.9 ms to 297.3 ms (n = 246)",
              "slow": 195.7,
              "slowKind": "95th percentile",
              "origin": "Jev live run: client wall time per call"
            },
            {
              "step": "OpenAI API · GPT-6 Luna · none (first output, one-line answer)",
              "covers": "time to first useful output",
              "route": "OpenAI API",
              "n": 5,
              "p50": 823,
              "range": "506 ms to 1,371 ms (n = 5)",
              "slow": 1371,
              "slowKind": "slowest run",
              "origin": "Provider explorer receipts: first useful output, matched cohort, fixed exact reply"
            },
            {
              "step": "OpenAI API · GPT-6.1 Sol · low (first output, one-line answer)",
              "covers": "time to first useful output",
              "route": "OpenAI API",
              "n": 5,
              "p50": 873,
              "range": "835 ms to 1,742 ms (n = 5)",
              "slow": 1742,
              "slowKind": "slowest run",
              "origin": "Provider explorer receipts: first useful output, matched cohort, fixed exact reply"
            },
            {
              "step": "Claude Code · Claude Fable 5.1 (first output, 5 short tasks)",
              "covers": "time to first useful output",
              "route": "Claude Code CLI",
              "n": 15,
              "p50": 1196,
              "range": "947 ms to 7,902 ms (n = 15)",
              "slow": 7902,
              "slowKind": "slowest run",
              "origin": "Provider head-to-head receipts: first useful output, configuration with the lowest observed median"
            },
            {
              "step": "OpenAI API · GPT-6.1 Sol · high (first output, one-line answer)",
              "covers": "time to first useful output",
              "route": "OpenAI API",
              "n": 5,
              "p50": 1341,
              "range": "1,259 ms to 2,115 ms (n = 5)",
              "slow": 2115,
              "slowKind": "slowest run",
              "origin": "Provider explorer receipts: first useful output, matched cohort, fixed exact reply"
            },
            {
              "step": "Claude Code · Claude Haiku 4.5 (start-up, first model output)",
              "covers": "time to first useful output",
              "route": "Claude Code CLI, one-word prompt",
              "n": 5,
              "p50": 1461,
              "range": "1,206 ms to 2,308 ms (n = 5)",
              "slow": 2308,
              "slowKind": "slowest run",
              "origin": "Routing overhead run: CLI start-up, time to the first model output"
            },
            {
              "step": "Claude Sonnet 5.5 (router, effort low, via Claude Code)",
              "covers": "whole decision, request to answer",
              "route": "Claude Code CLI",
              "n": 82,
              "p50": 2598,
              "range": "1,993 ms to 5,583 ms (n = 82)",
              "slow": 4298,
              "slowKind": "95th percentile",
              "origin": "Routing run: median and interpolated p95; overhead extract: n and observed limits"
            },
            {
              "step": "Codex CLI · GPT-6 Luna · none (first output, one-line answer)",
              "covers": "time to first useful output",
              "route": "Codex CLI",
              "n": 5,
              "p50": 2787,
              "range": "2,461 ms to 3,418 ms (n = 5)",
              "slow": 3418,
              "slowKind": "slowest run",
              "origin": "Provider explorer receipts: first useful output, matched cohort, fixed exact reply"
            },
            {
              "step": "Codex CLI · GPT-6.1 Sol · low (first output, one-line answer)",
              "covers": "time to first useful output",
              "route": "Codex CLI",
              "n": 5,
              "p50": 3753,
              "range": "3,435 ms to 4,103 ms (n = 5)",
              "slow": 4103,
              "slowKind": "slowest run",
              "origin": "Provider explorer receipts: first useful output, matched cohort, fixed exact reply"
            },
            {
              "step": "Codex CLI · GPT-6.1 Sol · high (first output, one-line answer)",
              "covers": "time to first useful output",
              "route": "Codex CLI",
              "n": 5,
              "p50": 3786,
              "range": "3,366 ms to 4,296 ms (n = 5)",
              "slow": 4296,
              "slowKind": "slowest run",
              "origin": "Provider explorer receipts: first useful output, matched cohort, fixed exact reply"
            },
            {
              "step": "Codex CLI (start-up, first model output)",
              "covers": "time to first useful output",
              "route": "Codex CLI, one-word prompt",
              "n": 5,
              "p50": 5059,
              "range": "4,391 ms to 5,478 ms (n = 5)",
              "slow": 5478,
              "slowKind": "slowest run",
              "origin": "Routing overhead run: CLI start-up, time to the first model output"
            },
            {
              "step": "Claude Haiku 4.5 (router, thinking on, via Claude Code)",
              "covers": "whole decision, request to answer",
              "route": "Claude Code CLI",
              "n": 82,
              "p50": 12674,
              "range": "5,857 ms to 51.28 s (n = 82)",
              "slow": 34413,
              "slowKind": "95th percentile",
              "origin": "Routing run: median and interpolated p95; overhead extract: n and observed limits"
            }
          ]
        },
        {
          "id": "latency-budget-fits",
          "title": "Which steps fit each assumed budget (calculation)",
          "columns": [
            {
              "key": "step",
              "label": "Step",
              "unit": "text"
            },
            {
              "key": "b300",
              "label": "Fits 300 ms",
              "unit": "text"
            },
            {
              "key": "b800",
              "label": "Fits 800 ms",
              "unit": "text"
            },
            {
              "key": "b1500",
              "label": "Fits 1,500 ms",
              "unit": "text"
            }
          ],
          "rows": [
            {
              "step": "Deterministic routing policy (Agent, in process)",
              "b300": "median and slow end",
              "b800": "median and slow end",
              "b1500": "median and slow end"
            },
            {
              "step": "Rule-based System One decision (record write excluded)",
              "b300": "median and slow end",
              "b800": "median and slow end",
              "b1500": "median and slow end"
            },
            {
              "step": "Jev 1.13 (TypeSafe, direct HTTPS)",
              "b300": "median and slow end",
              "b800": "median and slow end",
              "b1500": "median and slow end"
            },
            {
              "step": "OpenAI API · GPT-6 Luna · none (first output, one-line answer)",
              "b300": "neither",
              "b800": "neither",
              "b1500": "median and slow end"
            },
            {
              "step": "OpenAI API · GPT-6.1 Sol · low (first output, one-line answer)",
              "b300": "neither",
              "b800": "neither",
              "b1500": "median only"
            },
            {
              "step": "Claude Code · Claude Fable 5.1 (first output, 5 short tasks)",
              "b300": "neither",
              "b800": "neither",
              "b1500": "median only"
            },
            {
              "step": "OpenAI API · GPT-6.1 Sol · high (first output, one-line answer)",
              "b300": "neither",
              "b800": "neither",
              "b1500": "median only"
            },
            {
              "step": "Claude Code · Claude Haiku 4.5 (start-up, first model output)",
              "b300": "neither",
              "b800": "neither",
              "b1500": "median only"
            },
            {
              "step": "Claude Sonnet 5.5 (router, effort low, via Claude Code)",
              "b300": "neither",
              "b800": "neither",
              "b1500": "neither"
            },
            {
              "step": "Codex CLI · GPT-6 Luna · none (first output, one-line answer)",
              "b300": "neither",
              "b800": "neither",
              "b1500": "neither"
            },
            {
              "step": "Codex CLI · GPT-6.1 Sol · low (first output, one-line answer)",
              "b300": "neither",
              "b800": "neither",
              "b1500": "neither"
            },
            {
              "step": "Codex CLI · GPT-6.1 Sol · high (first output, one-line answer)",
              "b300": "neither",
              "b800": "neither",
              "b1500": "neither"
            },
            {
              "step": "Codex CLI (start-up, first model output)",
              "b300": "neither",
              "b800": "neither",
              "b1500": "neither"
            },
            {
              "step": "Claude Haiku 4.5 (router, thinking on, via Claude Code)",
              "b300": "neither",
              "b800": "neither",
              "b1500": "neither"
            }
          ]
        },
        {
          "id": "latency-budget-shares",
          "title": "Share of each assumed budget that one step uses (calculation)",
          "columns": [
            {
              "key": "step",
              "label": "Step",
              "unit": "text"
            },
            {
              "key": "m300",
              "label": "300 ms: median",
              "unit": "text"
            },
            {
              "key": "s300",
              "label": "300 ms: slow end",
              "unit": "text"
            },
            {
              "key": "m800",
              "label": "800 ms: median",
              "unit": "text"
            },
            {
              "key": "s800",
              "label": "800 ms: slow end",
              "unit": "text"
            },
            {
              "key": "m1500",
              "label": "1,500 ms: median",
              "unit": "text"
            },
            {
              "key": "s1500",
              "label": "1,500 ms: slow end",
              "unit": "text"
            }
          ],
          "rows": [
            {
              "step": "Deterministic routing policy (Agent, in process)",
              "m300": "under 0.1%",
              "s300": "under 0.1%",
              "m800": "under 0.1%",
              "s800": "under 0.1%",
              "m1500": "under 0.1%",
              "s1500": "under 0.1%"
            },
            {
              "step": "Rule-based System One decision (record write excluded)",
              "m300": "0.3%",
              "s300": "0.7%",
              "m800": "0.1%",
              "s800": "0.3%",
              "m1500": "under 0.1%",
              "s1500": "0.1%"
            },
            {
              "step": "Jev 1.13 (TypeSafe, direct HTTPS)",
              "m300": "46%",
              "s300": "65%",
              "m800": "17%",
              "s800": "24%",
              "m1500": "9.1%",
              "s1500": "13%"
            },
            {
              "step": "OpenAI API · GPT-6 Luna · none (first output, one-line answer)",
              "m300": "274%",
              "s300": "457%",
              "m800": "103%",
              "s800": "171%",
              "m1500": "55%",
              "s1500": "91%"
            },
            {
              "step": "OpenAI API · GPT-6.1 Sol · low (first output, one-line answer)",
              "m300": "291%",
              "s300": "581%",
              "m800": "109%",
              "s800": "218%",
              "m1500": "58%",
              "s1500": "116%"
            },
            {
              "step": "Claude Code · Claude Fable 5.1 (first output, 5 short tasks)",
              "m300": "399%",
              "s300": "2634%",
              "m800": "150%",
              "s800": "988%",
              "m1500": "80%",
              "s1500": "527%"
            },
            {
              "step": "OpenAI API · GPT-6.1 Sol · high (first output, one-line answer)",
              "m300": "447%",
              "s300": "705%",
              "m800": "168%",
              "s800": "264%",
              "m1500": "89%",
              "s1500": "141%"
            },
            {
              "step": "Claude Code · Claude Haiku 4.5 (start-up, first model output)",
              "m300": "487%",
              "s300": "769%",
              "m800": "183%",
              "s800": "289%",
              "m1500": "97%",
              "s1500": "154%"
            },
            {
              "step": "Claude Sonnet 5.5 (router, effort low, via Claude Code)",
              "m300": "866%",
              "s300": "1433%",
              "m800": "325%",
              "s800": "537%",
              "m1500": "173%",
              "s1500": "287%"
            },
            {
              "step": "Codex CLI · GPT-6 Luna · none (first output, one-line answer)",
              "m300": "929%",
              "s300": "1139%",
              "m800": "348%",
              "s800": "427%",
              "m1500": "186%",
              "s1500": "228%"
            },
            {
              "step": "Codex CLI · GPT-6.1 Sol · low (first output, one-line answer)",
              "m300": "1251%",
              "s300": "1368%",
              "m800": "469%",
              "s800": "513%",
              "m1500": "250%",
              "s1500": "274%"
            },
            {
              "step": "Codex CLI · GPT-6.1 Sol · high (first output, one-line answer)",
              "m300": "1262%",
              "s300": "1432%",
              "m800": "473%",
              "s800": "537%",
              "m1500": "252%",
              "s1500": "286%"
            },
            {
              "step": "Codex CLI (start-up, first model output)",
              "m300": "1686%",
              "s300": "1826%",
              "m800": "632%",
              "s800": "685%",
              "m1500": "337%",
              "s1500": "365%"
            },
            {
              "step": "Claude Haiku 4.5 (router, thinking on, via Claude Code)",
              "m300": "4225%",
              "s300": "11471%",
              "m800": "1584%",
              "s800": "4302%",
              "m1500": "845%",
              "s1500": "2294%"
            }
          ]
        },
        {
          "id": "latency-budget-in-sequence",
          "title": "How many runs of one step fit back to back in each budget (calculation, counts stop at 1,000)",
          "columns": [
            {
              "key": "step",
              "label": "Step",
              "unit": "text"
            },
            {
              "key": "m300",
              "label": "300 ms: at the median",
              "unit": "count"
            },
            {
              "key": "s300",
              "label": "300 ms: at the slow end",
              "unit": "count"
            },
            {
              "key": "m800",
              "label": "800 ms: at the median",
              "unit": "count"
            },
            {
              "key": "s800",
              "label": "800 ms: at the slow end",
              "unit": "count"
            },
            {
              "key": "m1500",
              "label": "1,500 ms: at the median",
              "unit": "count"
            },
            {
              "key": "s1500",
              "label": "1,500 ms: at the slow end",
              "unit": "count"
            }
          ],
          "rows": [
            {
              "step": "Deterministic routing policy (Agent, in process)",
              "m300": 1000,
              "s300": 1000,
              "m800": 1000,
              "s800": 1000,
              "m1500": 1000,
              "s1500": 1000
            },
            {
              "step": "Rule-based System One decision (record write excluded)",
              "m300": 300,
              "s300": 150,
              "m800": 800,
              "s800": 400,
              "m1500": 1000,
              "s1500": 750
            },
            {
              "step": "Jev 1.13 (TypeSafe, direct HTTPS)",
              "m300": 2,
              "s300": 1,
              "m800": 5,
              "s800": 4,
              "m1500": 10,
              "s1500": 7
            },
            {
              "step": "OpenAI API · GPT-6 Luna · none (first output, one-line answer)",
              "m300": 0,
              "s300": 0,
              "m800": 0,
              "s800": 0,
              "m1500": 1,
              "s1500": 1
            },
            {
              "step": "OpenAI API · GPT-6.1 Sol · low (first output, one-line answer)",
              "m300": 0,
              "s300": 0,
              "m800": 0,
              "s800": 0,
              "m1500": 1,
              "s1500": 0
            },
            {
              "step": "Claude Code · Claude Fable 5.1 (first output, 5 short tasks)",
              "m300": 0,
              "s300": 0,
              "m800": 0,
              "s800": 0,
              "m1500": 1,
              "s1500": 0
            },
            {
              "step": "OpenAI API · GPT-6.1 Sol · high (first output, one-line answer)",
              "m300": 0,
              "s300": 0,
              "m800": 0,
              "s800": 0,
              "m1500": 1,
              "s1500": 0
            },
            {
              "step": "Claude Code · Claude Haiku 4.5 (start-up, first model output)",
              "m300": 0,
              "s300": 0,
              "m800": 0,
              "s800": 0,
              "m1500": 1,
              "s1500": 0
            },
            {
              "step": "Claude Sonnet 5.5 (router, effort low, via Claude Code)",
              "m300": 0,
              "s300": 0,
              "m800": 0,
              "s800": 0,
              "m1500": 0,
              "s1500": 0
            },
            {
              "step": "Codex CLI · GPT-6 Luna · none (first output, one-line answer)",
              "m300": 0,
              "s300": 0,
              "m800": 0,
              "s800": 0,
              "m1500": 0,
              "s1500": 0
            },
            {
              "step": "Codex CLI · GPT-6.1 Sol · low (first output, one-line answer)",
              "m300": 0,
              "s300": 0,
              "m800": 0,
              "s800": 0,
              "m1500": 0,
              "s1500": 0
            },
            {
              "step": "Codex CLI · GPT-6.1 Sol · high (first output, one-line answer)",
              "m300": 0,
              "s300": 0,
              "m800": 0,
              "s800": 0,
              "m1500": 0,
              "s1500": 0
            },
            {
              "step": "Codex CLI (start-up, first model output)",
              "m300": 0,
              "s300": 0,
              "m800": 0,
              "s800": 0,
              "m1500": 0,
              "s1500": 0
            },
            {
              "step": "Claude Haiku 4.5 (router, thinking on, via Claude Code)",
              "m300": 0,
              "s300": 0,
              "m800": 0,
              "s800": 0,
              "m1500": 0,
              "s1500": 0
            }
          ]
        },
        {
          "id": "latency-budget-router-plus-model",
          "title": "A router in front of a model: the two times added (calculation)",
          "columns": [
            {
              "key": "router",
              "label": "Router (decision)",
              "unit": "text"
            },
            {
              "key": "model",
              "label": "Model (first output)",
              "unit": "text"
            },
            {
              "key": "p50",
              "label": "Sum of medians",
              "unit": "ms"
            },
            {
              "key": "slow",
              "label": "Sum of slow ends",
              "unit": "ms"
            },
            {
              "key": "b300",
              "label": "Fits 300 ms",
              "unit": "text"
            },
            {
              "key": "b800",
              "label": "Fits 800 ms",
              "unit": "text"
            },
            {
              "key": "b1500",
              "label": "Fits 1,500 ms",
              "unit": "text"
            },
            {
              "key": "left1500",
              "label": "Left of 1,500 ms at the median (negative: over)",
              "unit": "ms"
            },
            {
              "key": "leftSlow1500",
              "label": "Left of 1,500 ms at the slow end (negative: over)",
              "unit": "ms"
            }
          ],
          "rows": [
            {
              "router": "Deterministic routing policy (Agent, in process)",
              "model": "OpenAI API · GPT-6 Luna · none (first output, one-line answer)",
              "p50": 823,
              "slow": 1371,
              "b300": "neither",
              "b800": "neither",
              "b1500": "median and slow end",
              "left1500": 677,
              "leftSlow1500": 129
            },
            {
              "router": "Deterministic routing policy (Agent, in process)",
              "model": "OpenAI API · GPT-6.1 Sol · low (first output, one-line answer)",
              "p50": 873,
              "slow": 1742,
              "b300": "neither",
              "b800": "neither",
              "b1500": "median only",
              "left1500": 627,
              "leftSlow1500": -242
            },
            {
              "router": "Jev 1.13 (TypeSafe, direct HTTPS)",
              "model": "OpenAI API · GPT-6 Luna · none (first output, one-line answer)",
              "p50": 959.5,
              "slow": 1566.7,
              "b300": "neither",
              "b800": "neither",
              "b1500": "median only",
              "left1500": 540.5,
              "leftSlow1500": -66.7
            },
            {
              "router": "Jev 1.13 (TypeSafe, direct HTTPS)",
              "model": "OpenAI API · GPT-6.1 Sol · low (first output, one-line answer)",
              "p50": 1009.5,
              "slow": 1937.7,
              "b300": "neither",
              "b800": "neither",
              "b1500": "median only",
              "left1500": 490.5,
              "leftSlow1500": -437.7
            },
            {
              "router": "Claude Sonnet 5.5 (router, effort low, via Claude Code)",
              "model": "OpenAI API · GPT-6 Luna · none (first output, one-line answer)",
              "p50": 3421,
              "slow": 5669,
              "b300": "neither",
              "b800": "neither",
              "b1500": "neither",
              "left1500": -1921,
              "leftSlow1500": -4169
            },
            {
              "router": "Claude Sonnet 5.5 (router, effort low, via Claude Code)",
              "model": "OpenAI API · GPT-6.1 Sol · low (first output, one-line answer)",
              "p50": 3471,
              "slow": 6040,
              "b300": "neither",
              "b800": "neither",
              "b1500": "neither",
              "left1500": -1971,
              "leftSlow1500": -4540
            }
          ]
        }
      ],
      "related": [
        "routing-overhead",
        "cli-model-latency-tokens"
      ]
    },
    {
      "slug": "routing-at-scale",
      "title": "What does routing a million AI requests a day cost? A calculation from measured runs",
      "seoTitle": "LLM router cost at scale: 1 million decisions a day",
      "description": "A calculation from measured runs: what 10,000 to 10 million routing decisions a day cost with rules, Jev and Claude, with median/p95 time scenarios.",
      "question": "At 10,000 to 10 million routing decisions a day, what does each router cost per day, what in-flight and waiting scenarios do median and p95 times give?",
      "answer": "This is a calculation, not a run. We multiplied measured numbers and made no model call. At 1 million routing decisions a day, the rule-based policy costs $0 and Jev costs $33.70. A Sonnet 5.5 router costs $7,324 and a Haiku 4.5 router costs $8,924. Cost samples: Jev n = 82, Sonnet n = 82, Haiku n = 82. Jev scales reported run cost. Claude uses list prices and reported tokens. These calculations fix the token mix and have no statistical cost bounds. At 10 million decisions a day, Jev costs $337.00, Sonnet $73,240 and Haiku $89,240. At 1 million a day, that is 11.57 decisions per second. At its median time, the Sonnet scenario gives about 30.1 decisions in flight at once (49.7 at its 95th percentile time). At its median time, the Haiku scenario gives about 146.7 (398.3). At its median time, the Jev scenario gives about 1.58 (2.27), timed live over direct HTTPS. If every decision took its median time, Sonnet would add 721.7 hours of waiting a day and Jev would add 37.9 hours. The policy median-time scenario adds 1.42 s. Say the router decides only the System One decisions (14.1% of model calls). The Sonnet cost at 1 million calls a day then falls from $7,324 to $1,036. Time samples: policy n = 20000, Sonnet n = 82, Haiku n = 82, Jev n = 246. These are median/p95 scenarios, not measured totals or confidence intervals. We did not measure rate limits, load or accuracy.",
      "date": "2026-10-06",
      "updated": "2026-10-06",
      "tags": [
        "thought-experiment",
        "calculation",
        "routing",
        "llm-router",
        "jev",
        "latency",
        "cost",
        "scale",
        "littles-law"
      ],
      "method": [
        "This is a calculation, not a run. We made no new model call and ran nothing at scale. Every input is a number from a recorded run in this dataset. The volumes (10,000, 100,000, 1 million, 10 million decisions a day) are parameters, not measurements.",
        "Jev cost per 1,000 is a calculation from the routing study’s provider-reported total cost and counted calls. Claude cost comes from its token totals and list prices, with one-hour cache writes. Jev costs $0.0337 per 1,000, a calculation from the reported production run cost (n = 82). \n\nSonnet 5.5 at effort low and Haiku 4.5 ran through the Claude Code CLI. Their cost is list price × the tokens the CLI reported (Sonnet n = 82, Haiku n = 82). The policy makes no model call, so it costs $0. The live Jev run gives $0.0337 per 1,000, a calculation from reported input tokens and the published price. The two figures agree at this precision.",
        "Claude median and p95 times come from the routing receipts; we checked them against the per-call rows. The overhead summary uses different quantiles for those calls, so we do not mix its median or p95 into this calculation. Run ranges and policy times come from the overhead results. The policy is the production routing decision, timed in process over 20,000 decisions (median 1.42 µs, 95th percentile 2.33 µs, maximum 2538.21 µs; minimum unavailable). \n\nSonnet 5.5 and Haiku 4.5 are wall time per call through the CLI, one call at a time, from the recorded routing runs (Sonnet n = 82, Haiku n = 82). Jev is the live run: 246 calls (82 typed decisions × 3 repetitions) over direct HTTPS from one Mac, one call at a time, as client wall time.",
        "Daily cost = decisions a day × cost per decision. Decisions per second = decisions a day ÷ 86,400. Decisions in flight = decisions per second × time per decision in seconds (median/p95 scenarios; Little’s law requires the mean). The median time gives the bar. The 95th percentile time gives the whisker. \n\nWaiting hours a day = decisions a day × time per decision ÷ 3,600, if each request waits for its decision. Cost per year = cost per day × 365.",
        "The scenarios use 1,000,000 model calls a day. In the first, a router decides every call. In the second, it decides only the System One decisions: 7 ÷ 49.5 = 14.1% of calls. That ratio uses the medians of 48 recorded bench runs. \n\nRouting was off in those runs, so each model call counts as one decision a router could make. In the third, the policy decides every call in process.",
        "Policy capacity is the measured 469,409 decisions per second in one process, set against the rate each volume needs. The measurement leaves out the database reads and the decision-record write of the production decision."
      ],
      "caveats": [
        "The routing and live Jev protocols predate the first counted calls by file birth time. The overhead protocol predates its microbenchmark output. Claude ran 82 calls per router, below its 120-call cap. Jev ran 246 counted calls, below its 300-call cap. One Haiku pilot and one Jev probe are excluded. All counted calls completed without call errors.\n\nThe policy microbenchmark used 5,000 warm-up iterations and 64 synthetic contexts; the minimum and individual timing samples are unavailable, so we cannot rebuild its quantiles or full range. No separate validator-control receipt or Sonnet pilot is retained.",
        "The source accuracy sets hit a ceiling in some purposes. These tuned cases do not establish equal quality on new traffic. The same 82 Jev cases repeat three times; 246 latency calls are not 246 distinct tasks.",
        "Nothing ran at scale. We did not measure rate limits, queueing or slowdown under load. The Claude routers ran one call at a time through a CLI, and the live Jev run also ran one call at a time. The calculation does not show that a provider accepts this many decisions a day, or this many at once.",
        "The routes differ. The policy runs in process. We timed Jev over direct HTTPS from one Mac on a home network, so its time includes the network round trip. The Claude routers ran through the Claude Code CLI. The CLI adds time (a median 975 ms for Sonnet and 1,701 ms for Haiku) and its own prompt to every call. \n\nWe did not time a direct API call, so the Claude rows show the CLI route and not the best a Claude router can do.",
        "The Claude costs are calculations on subscription calls, not invoices. Sonnet's receipts report one-hour cache writes. The original routing estimate used five-minute write rates and gave $5.00 per 1,000. This study uses the one-hour rate and gives $7.32.\n\nThe CLI estimate is $7.32 per 1,000. Mean output tokens: Haiku 1,419, Sonnet 107. Haiku used default thinking; this comparison does not isolate its effect.",
        "Little’s law uses the mean time per decision. The public CLI extracts contain the median and p95, but no mean. Neither value bounds the mean. These are assumed-time scenarios, not average concurrency, actual waiting totals or capacity guarantees. For Jev, the mean (142.1 ms) is 4.1% above the median (calculation). \n\nThe waiting total has the same limit. Each whisker runs from the median calculation to the 95th percentile calculation. It is not a confidence interval.",
        "A steady rate over 24 hours is an assumption. Real traffic has peaks. To size a peak, multiply the in-flight figures by the peak rate ÷ the average rate.",
        "We did not score accuracy here. The policy picks a model and effort from typed signals. Jev and the Claude routers answer typed System One decisions. The routing study scores the three model routers on 82 decisions. We tuned the case sets against Jev’s answers, so Jev has a home advantage. A cheap router that is wrong costs more in the work it misroutes, and this calculation does not price that.",
        "The System One share comes from 48 bench runs of one agent. The ratio of medians is 14.1%. Pooled over the runs, it is 17.0%, which raises the System One scenario cost by 20% (calculation). Your traffic may differ. The work cost per call is Agent’s notional list-price cost on Sonnet 5.5, not a bill.",
        "The policy’s 1.42 µs median leaves out database reads and the decision-record write, and it needs a server to run on. The calculation prices model calls only. It does not price servers, retries or engineering time.",
        "The samples are small. Each Claude router has 82 decisions (one sample per case). The Jev cost has 82, the live Jev time has 246 calls and the policy time has 20,000. A cost per decision is one estimate with no interval. The projection fixes that estimate and the token mix; it has no statistical cost bounds.",
        "We did not isolate the Mac that ran the timed calls. Other jobs, such as local model servers, may have run at the same time, so a wall time may include contention."
      ],
      "sourceIds": [
        "calc-routing-at-scale",
        "agent-routing-overhead",
        "agent-routing",
        "price-anthropic",
        "price-jev",
        "agent-jev-live"
      ],
      "stats": [
        {
          "id": "routing-at-scale-cost-per-1000-jev",
          "label": "Cost per 1,000 routing decisions, Jev (calculation)",
          "value": 0.0337,
          "unit": "usd",
          "display": "$0.0337",
          "n": 82,
          "note": "Calculation: provider-reported total cost ÷ counted calls × 1,000. This fixes the recorded token mix; no cost interval is available."
        },
        {
          "id": "routing-at-scale-cost-per-1000-sonnet",
          "label": "Cost per 1,000 routing decisions, Sonnet 5.5 router (calculation)",
          "value": 7.324,
          "unit": "usd",
          "display": "$7.32",
          "n": 82,
          "note": "A list-price calculation on the tokens the CLI reported. The calls ran on a subscription."
        },
        {
          "id": "routing-at-scale-cost-per-1000-haiku",
          "label": "Cost per 1,000 routing decisions, Haiku 4.5 router (calculation)",
          "value": 8.924,
          "unit": "usd",
          "display": "$8.92",
          "n": 82,
          "note": "A list-price calculation on the tokens the CLI reported. The calls ran on a subscription."
        },
        {
          "id": "routing-at-scale-system-one-share",
          "label": "System One decisions as a share of model calls (ratio of the two medians, calculation)",
          "value": 0.1414,
          "unit": "rate",
          "display": "14.1%",
          "n": 48,
          "note": "7 ÷ 49.5. A ratio of medians, not a sample proportion, so it carries no interval."
        },
        {
          "id": "routing-at-scale-rate-1m",
          "label": "Decisions per second at 1 million a day (calculation)",
          "value": 11.5741,
          "unit": "count",
          "display": "11.57 per second",
          "note": "A steady rate over 24 hours."
        },
        {
          "id": "routing-at-scale-cost-1m-policy",
          "label": "Daily cost at 1 million decisions, rule-based policy (calculation)",
          "value": 0,
          "unit": "usd",
          "display": "$0",
          "n": 20000,
          "note": "$0 a year at a flat volume."
        },
        {
          "id": "routing-at-scale-cost-1m-jev",
          "label": "Daily cost at 1 million decisions, Jev (calculation)",
          "value": 33.7,
          "unit": "usd",
          "display": "$33.70",
          "n": 82,
          "note": "$12,301 a year at a flat volume."
        },
        {
          "id": "routing-at-scale-cost-1m-sonnet",
          "label": "Daily cost at 1 million decisions, Sonnet 5.5 router (calculation)",
          "value": 7324,
          "unit": "usd",
          "display": "$7,324",
          "n": 82,
          "note": "$2,673,260 a year at a flat volume."
        },
        {
          "id": "routing-at-scale-cost-1m-haiku",
          "label": "Daily cost at 1 million decisions, Haiku 4.5 router (calculation)",
          "value": 8924,
          "unit": "usd",
          "display": "$8,924",
          "n": 82,
          "note": "$3,257,260 a year at a flat volume."
        },
        {
          "id": "routing-at-scale-flight-1m-jev",
          "label": "Decisions in flight at 1 million a day, Jev (calculation)",
          "value": 1.5799,
          "unit": "calls",
          "display": "1.58 (2.27 at the 95th percentile time)",
          "n": 246,
          "note": "Decisions per second × assumed median or p95 time. These are scenarios, not actual mean concurrency or confidence bounds."
        },
        {
          "id": "routing-at-scale-flight-1m-sonnet",
          "label": "Decisions in flight at 1 million a day, Sonnet 5.5 router (calculation)",
          "value": 30.069,
          "unit": "calls",
          "display": "30.1 (49.7 at the 95th percentile time)",
          "n": 82,
          "note": "Decisions per second × assumed median or p95 time. These are scenarios, not actual mean concurrency or confidence bounds."
        },
        {
          "id": "routing-at-scale-flight-1m-haiku",
          "label": "Decisions in flight at 1 million a day, Haiku 4.5 router (calculation)",
          "value": 146.69,
          "unit": "calls",
          "display": "146.7 (398.3 at the 95th percentile time)",
          "n": 82,
          "note": "Decisions per second × assumed median or p95 time. These are scenarios, not actual mean concurrency or confidence bounds."
        },
        {
          "id": "routing-at-scale-hours-1m-policy",
          "label": "Waiting hours a day at 1 million decisions, rule-based policy (calculation)",
          "value": 0.00039444,
          "unit": "count",
          "display": "1.42 s in total",
          "n": 20000,
          "note": "Decisions × assumed median or p95 time. Actual total waiting requires the mean; these are not confidence bounds."
        },
        {
          "id": "routing-at-scale-hours-1m-jev",
          "label": "Waiting hours a day at 1 million decisions, Jev (calculation)",
          "value": 37.917,
          "unit": "count",
          "display": "37.9 hours (54.4 at the 95th percentile time)",
          "n": 246,
          "note": "Decisions × assumed median or p95 time. Actual total waiting requires the mean; these are not confidence bounds."
        },
        {
          "id": "routing-at-scale-hours-1m-sonnet",
          "label": "Waiting hours a day at 1 million decisions, Sonnet 5.5 router (calculation)",
          "value": 721.67,
          "unit": "count",
          "display": "721.7 hours (1,194 at the 95th percentile time)",
          "n": 82,
          "note": "Decisions × assumed median or p95 time. Actual total waiting requires the mean; these are not confidence bounds."
        }
      ],
      "charts": [
        {
          "id": "routing-at-scale-daily-cost",
          "title": "Daily cost of routing at 10,000 to 10 million decisions a day (calculation)",
          "subtitle": "Decisions a day × cost per decision, USD per day, every decision routed. The rule-based policy costs $0 and is not on the log axis",
          "kind": "grouped-bar",
          "unit": "usd",
          "yLabel": "USD per day",
          "series": [
            {
              "name": "Jev 1.13 (TypeSafe)",
              "points": [
                {
                  "label": "10,000 a day",
                  "value": 0.337,
                  "n": 82
                },
                {
                  "label": "100,000 a day",
                  "value": 3.37,
                  "n": 82
                },
                {
                  "label": "1 million a day",
                  "value": 33.7,
                  "n": 82
                },
                {
                  "label": "10 million a day",
                  "value": 337,
                  "n": 82
                }
              ]
            },
            {
              "name": "Claude Sonnet 5.5 (low) · Claude Code",
              "points": [
                {
                  "label": "10,000 a day",
                  "value": 73.24,
                  "n": 82
                },
                {
                  "label": "100,000 a day",
                  "value": 732.4,
                  "n": 82
                },
                {
                  "label": "1 million a day",
                  "value": 7324,
                  "n": 82
                },
                {
                  "label": "10 million a day",
                  "value": 73240,
                  "n": 82
                }
              ]
            },
            {
              "name": "Claude Haiku 4.5 · Claude Code",
              "points": [
                {
                  "label": "10,000 a day",
                  "value": 89.24,
                  "n": 82
                },
                {
                  "label": "100,000 a day",
                  "value": 892.4,
                  "n": 82
                },
                {
                  "label": "1 million a day",
                  "value": 8924,
                  "n": 82
                },
                {
                  "label": "10 million a day",
                  "value": 89240,
                  "n": 82
                }
              ]
            }
          ],
          "note": "This is a calculation, not a run. Daily cost = decisions a day × cost per decision. Cost per 1,000 decisions: Jev $0.0337 (calculation from provider-reported run cost, n = 82), Sonnet 5.5 $7.32, Haiku 4.5 $8.92. The Claude figures use one-hour cache-write rates and list price × the tokens the CLI reported (Sonnet n = 82, Haiku n = 82). \n\nThe rule-based policy makes no model call, so it costs $0 at every volume and has no bar: a log axis cannot show zero. The chart prices model calls only, not servers, retries or the work itself.",
          "sourceIds": [
            "calc-routing-at-scale",
            "agent-routing-overhead",
            "agent-routing",
            "price-anthropic",
            "price-jev"
          ]
        },
        {
          "id": "routing-at-scale-in-flight",
          "title": "In-flight scenarios at 10,000 to 10 million a day (calculation)",
          "subtitle": "Decisions per second × assumed time per decision; bar = median time, whisker = the same calculation at the 95th percentile. The in-process policy is in the table",
          "kind": "grouped-bar",
          "unit": "calls",
          "yLabel": "In-flight scenarios at median / p95 time",
          "whisker": "p50-p95",
          "series": [
            {
              "name": "Jev 1.13 (TypeSafe)",
              "points": [
                {
                  "label": "10,000 a day",
                  "value": 0.015799,
                  "lo": 0.015799,
                  "hi": 0.02265,
                  "n": 246
                },
                {
                  "label": "100,000 a day",
                  "value": 0.15799,
                  "lo": 0.15799,
                  "hi": 0.2265,
                  "n": 246
                },
                {
                  "label": "1 million a day",
                  "value": 1.5799,
                  "lo": 1.5799,
                  "hi": 2.265,
                  "n": 246
                },
                {
                  "label": "10 million a day",
                  "value": 15.799,
                  "lo": 15.799,
                  "hi": 22.65,
                  "n": 246
                }
              ]
            },
            {
              "name": "Claude Sonnet 5.5 (low) · Claude Code",
              "points": [
                {
                  "label": "10,000 a day",
                  "value": 0.30069,
                  "lo": 0.30069,
                  "hi": 0.49745,
                  "n": 82
                },
                {
                  "label": "100,000 a day",
                  "value": 3.0069,
                  "lo": 3.0069,
                  "hi": 4.9745,
                  "n": 82
                },
                {
                  "label": "1 million a day",
                  "value": 30.069,
                  "lo": 30.069,
                  "hi": 49.745,
                  "n": 82
                },
                {
                  "label": "10 million a day",
                  "value": 300.69,
                  "lo": 300.69,
                  "hi": 497.45,
                  "n": 82
                }
              ]
            },
            {
              "name": "Claude Haiku 4.5 · Claude Code",
              "points": [
                {
                  "label": "10,000 a day",
                  "value": 1.4669,
                  "lo": 1.4669,
                  "hi": 3.983,
                  "n": 82
                },
                {
                  "label": "100,000 a day",
                  "value": 14.669,
                  "lo": 14.669,
                  "hi": 39.83,
                  "n": 82
                },
                {
                  "label": "1 million a day",
                  "value": 146.69,
                  "lo": 146.69,
                  "hi": 398.3,
                  "n": 82
                },
                {
                  "label": "10 million a day",
                  "value": 1466.9,
                  "lo": 1466.9,
                  "hi": 3983,
                  "n": 82
                }
              ]
            }
          ],
          "note": "This is a calculation, not a run. In flight = decisions a day ÷ 86,400 × time per decision, at a steady rate over 24 hours. The bar uses the median time. The whisker uses the 95th percentile time and is not a confidence interval. These are assumed-time scenarios. Actual average concurrency requires the mean time. \n\nThe inputs table lists the times. Jev ran over direct HTTPS and the Claude routers ran through a CLI, so the routes differ. Traffic has peaks, so multiply by the peak rate ÷ the average rate to size a peak. \n\nThe rule-based policy has no row. It runs in process, so at 1 million decisions a day the median-time scenario gives 0.000016 decisions in flight, a small share of one decision and not a connection. Labels round small values. The grid table lists every value.",
          "sourceIds": [
            "calc-routing-at-scale",
            "agent-routing-overhead",
            "agent-routing",
            "agent-jev-live"
          ]
        },
        {
          "id": "routing-at-scale-waiting-hours",
          "title": "Waiting scenarios per day at 1 million decisions a day (calculation)",
          "subtitle": "Decisions × time per decision, in hours of waiting added across all requests; whisker = the same calculation at the 95th percentile",
          "kind": "bar",
          "unit": "count",
          "yLabel": "Waiting hours at assumed median / p95 time",
          "whisker": "p50-p95",
          "series": [
            {
              "name": "Hours of waiting per day",
              "points": [
                {
                  "label": "Jev 1.13 (TypeSafe)",
                  "value": 37.917,
                  "lo": 37.917,
                  "hi": 54.361,
                  "n": 246
                },
                {
                  "label": "Claude Sonnet 5.5 (low) · Claude Code",
                  "value": 721.67,
                  "lo": 721.67,
                  "hi": 1193.9,
                  "n": 82
                },
                {
                  "label": "Claude Haiku 4.5 · Claude Code",
                  "value": 3520.6,
                  "lo": 3520.6,
                  "hi": 9559.2,
                  "n": 82
                }
              ]
            }
          ],
          "note": "This is a calculation, not a run. Hours of waiting a day = decisions a day × time per decision ÷ 3,600, if each request waits for its decision. The bar uses the median time. The whisker uses the 95th percentile time and is not a confidence interval. Total waiting equals decisions × the mean time. The public extracts do not contain the mean for the CLI routers, so the true total is not known. \n\nThe rule-based policy has no bar: at this volume the median-time scenario gives 1.42 s in total. A decision outside the request’s critical path may add no waiting for that request.",
          "sourceIds": [
            "calc-routing-at-scale",
            "agent-routing-overhead",
            "agent-routing",
            "agent-jev-live"
          ]
        },
        {
          "id": "routing-at-scale-scenarios",
          "title": "Daily cost at 1 million model calls a day: route every call or only System One decisions (calculation)",
          "subtitle": "Every call is one decision; System One decisions are 7 of 49.5 model calls per task in the recorded runs",
          "kind": "grouped-bar",
          "unit": "usd",
          "yLabel": "USD per day",
          "series": [
            {
              "name": "Every model call routed",
              "points": [
                {
                  "label": "Deterministic routing policy",
                  "value": 0,
                  "n": 20000
                },
                {
                  "label": "Jev 1.13 (TypeSafe)",
                  "value": 33.7,
                  "n": 82
                },
                {
                  "label": "Claude Sonnet 5.5 (low) · Claude Code",
                  "value": 7324,
                  "n": 82
                },
                {
                  "label": "Claude Haiku 4.5 · Claude Code",
                  "value": 8924,
                  "n": 82
                }
              ]
            },
            {
              "name": "Only System One decisions routed (14.1% of calls)",
              "points": [
                {
                  "label": "Deterministic routing policy",
                  "value": 0,
                  "n": 20000
                },
                {
                  "label": "Jev 1.13 (TypeSafe)",
                  "value": 4.7657,
                  "n": 82
                },
                {
                  "label": "Claude Sonnet 5.5 (low) · Claude Code",
                  "value": 1035.7,
                  "n": 82
                },
                {
                  "label": "Claude Haiku 4.5 · Claude Code",
                  "value": 1262,
                  "n": 82
                }
              ]
            }
          ],
          "note": "This is a calculation, not a run. It assumes 1,000,000 model calls a day. Route every call: 1,000,000 decisions. Route only System One decisions: 141,414 decisions, which is 7 ÷ 49.5 = 14.1% of calls. \n\nSystem One decisions are the typed choices an agent asks a router to make. The 7 and the 49.5 are medians over 48 recorded bench runs. \n\nRouting was off in those runs, so each model call counts as one decision a router could make. Pooled over the runs, System One decisions are 17.0% of calls (419 of 2,467). The policy is the third scenario: it decides every call in process at no model cost.",
          "sourceIds": [
            "calc-routing-at-scale",
            "agent-routing-overhead",
            "agent-routing",
            "price-anthropic",
            "price-jev"
          ]
        }
      ],
      "tables": [
        {
          "id": "routing-at-scale-inputs",
          "title": "Recorded inputs and cost calculations (per decision)",
          "columns": [
            {
              "key": "router",
              "label": "Router",
              "unit": "text"
            },
            {
              "key": "costPer1000",
              "label": "Cost per 1,000 decisions",
              "unit": "usd"
            },
            {
              "key": "costBasis",
              "label": "Cost basis",
              "unit": "text"
            },
            {
              "key": "costN",
              "label": "n (cost)",
              "unit": "count"
            },
            {
              "key": "p50",
              "label": "Median time per decision",
              "unit": "ms"
            },
            {
              "key": "p95",
              "label": "95th percentile",
              "unit": "ms"
            },
            {
              "key": "min",
              "label": "Fastest call (range, not an interval)",
              "unit": "ms"
            },
            {
              "key": "max",
              "label": "Slowest call (range, not an interval)",
              "unit": "ms"
            },
            {
              "key": "latN",
              "label": "n (time)",
              "unit": "count"
            },
            {
              "key": "latBasis",
              "label": "Time basis",
              "unit": "text"
            }
          ],
          "rows": [
            {
              "router": "Deterministic routing policy",
              "costPer1000": 0,
              "costBasis": "no model call",
              "costN": 20000,
              "p50": 0.00142,
              "p95": 0.00233,
              "min": null,
              "max": 2.53821,
              "latN": 20000,
              "latBasis": "timed in process (microbenchmark, no database reads)"
            },
            {
              "router": "Jev 1.13 (TypeSafe)",
              "costPer1000": 0.0337,
              "costBasis": "provider-reported run cost ÷ calls × 1,000 (calculation)",
              "costN": 82,
              "p50": 136.5,
              "p95": 195.7,
              "min": 100.9,
              "max": 297.3,
              "latN": 246,
              "latBasis": "live run over direct HTTPS, one call at a time (client wall time)"
            },
            {
              "router": "Claude Sonnet 5.5 (low) · Claude Code",
              "costPer1000": 7.324,
              "costBasis": "list price × CLI-reported tokens (calculation)",
              "costN": 82,
              "p50": 2598,
              "p95": 4298,
              "min": 1993,
              "max": 5583,
              "latN": 82,
              "latBasis": "recorded routing study, through the Claude Code CLI, one call at a time"
            },
            {
              "router": "Claude Haiku 4.5 · Claude Code",
              "costPer1000": 8.924,
              "costBasis": "list price × CLI-reported tokens (calculation)",
              "costN": 82,
              "p50": 12674,
              "p95": 34413,
              "min": 5857,
              "max": 51278,
              "latN": 82,
              "latBasis": "recorded routing study, through the Claude Code CLI, one call at a time (default thinking on)"
            }
          ]
        },
        {
          "id": "routing-at-scale-formulas",
          "title": "The formulas",
          "columns": [
            {
              "key": "quantity",
              "label": "Quantity",
              "unit": "text"
            },
            {
              "key": "formula",
              "label": "Formula",
              "unit": "text"
            },
            {
              "key": "inputs",
              "label": "Inputs",
              "unit": "text"
            }
          ],
          "rows": [
            {
              "quantity": "Cost per day",
              "formula": "decisions a day × cost per 1,000 decisions ÷ 1,000",
              "inputs": "Routing receipts: Jev reported total cost ÷ calls × 1,000 (calculation); Claude tokensPerDecision × list prices, with one-hour cache writes"
            },
            {
              "quantity": "Decisions per second",
              "formula": "decisions a day ÷ 86,400",
              "inputs": "A steady rate over 24 hours"
            },
            {
              "quantity": "Decisions in flight",
              "formula": "decisions per second × assumed median or p95 time per decision in seconds",
              "inputs": "Routing-overhead results: policy latencyUs p50 and p95; Claude routing receipts latencyMs wallP50 and wallP95; live Jev run: latencyMs.calls median and p95"
            },
            {
              "quantity": "Waiting hours a day",
              "formula": "decisions a day × time per decision in seconds ÷ 3,600",
              "inputs": "The same times, if each request waits for its decision"
            },
            {
              "quantity": "System One scenario",
              "formula": "decisions = model calls a day × (System One decisions per task ÷ model calls per task) = × 7 ÷ 49.5",
              "inputs": "Routing-overhead results: perTask.systemOneDecisions.median and perTask.modelCalls.median"
            },
            {
              "quantity": "Policy share of one process",
              "formula": "decisions per second ÷ measured decisions per second",
              "inputs": "Routing-overhead results: policy decisionsPerSecond"
            },
            {
              "quantity": "Cost per year",
              "formula": "cost per day × 365",
              "inputs": "A flat daily volume"
            }
          ]
        },
        {
          "id": "routing-at-scale-context",
          "title": "Counts, inputs and scale checks behind the calculation",
          "columns": [
            {
              "key": "quantity",
              "label": "Quantity",
              "unit": "text"
            },
            {
              "key": "value",
              "label": "Value",
              "unit": "text"
            },
            {
              "key": "n",
              "label": "n",
              "unit": "count"
            },
            {
              "key": "note",
              "label": "Note",
              "unit": "text"
            }
          ],
          "rows": [
            {
              "quantity": "Cost per 1,000 routing decisions, rule-based policy",
              "value": "$0",
              "n": 20000,
              "note": "The policy makes no model call."
            },
            {
              "quantity": "Median time per routing decision, rule-based policy",
              "value": "1.42 µs (p95 2.33 µs); maximum 2538.21 µs (minimum unavailable)",
              "n": 20000,
              "note": "timed in process (microbenchmark, no database reads). The 95th percentile is a point on the time distribution. It is not a confidence interval. The run range, where shown, is not an interval. The minimum and individual timing samples are unavailable; we cannot rebuild the quantiles or full range."
            },
            {
              "quantity": "Median time per routing decision, Jev",
              "value": "136.5 ms (p95 195.7 ms); range 100.9 ms to 297.3 ms",
              "n": 246,
              "note": "live run over direct HTTPS, one call at a time (client wall time). The 95th percentile is a point on the time distribution. It is not a confidence interval. The run range, where shown, is not an interval."
            },
            {
              "quantity": "Median time per routing decision, Sonnet 5.5 router",
              "value": "2.60 s (p95 4.30 s); range 1.99 s to 5.58 s",
              "n": 82,
              "note": "recorded routing study, through the Claude Code CLI, one call at a time. The 95th percentile is a point on the time distribution. It is not a confidence interval. The run range, where shown, is not an interval."
            },
            {
              "quantity": "Median time per routing decision, Haiku 4.5 router",
              "value": "12.67 s (p95 34.41 s); range 5.86 s to 51.28 s",
              "n": 82,
              "note": "recorded routing study, through the Claude Code CLI, one call at a time (default thinking on). The 95th percentile is a point on the time distribution. It is not a confidence interval. The run range, where shown, is not an interval."
            },
            {
              "quantity": "Model calls per task, median of the recorded runs",
              "value": "49.5",
              "n": 48,
              "note": "Each model call counts as one decision a router could make. Recorded range: 13 to 73 calls; not an interval."
            },
            {
              "quantity": "System One decisions per task, median of the recorded runs",
              "value": "7",
              "n": 48,
              "note": "Recorded range: 2 to 24 decisions; not an interval."
            },
            {
              "quantity": "System One decisions as a share of model calls, pooled over the runs (calculation)",
              "value": "17.0% (419 of 2,467)",
              "n": 48,
              "note": "Total System One decisions ÷ total model calls over the runs with at least one call. It differs from the ratio of medians. The scenario chart uses the ratio of medians, as the routing-overhead study does."
            },
            {
              "quantity": "Recorded work cost per model call, median task cost ÷ median calls (calculation)",
              "value": "$0.0611",
              "n": 48,
              "note": "Notional list-price cost of Agent’s recorded bench work on Sonnet 5.5; the runs used a subscription."
            },
            {
              "quantity": "Decisions in flight at 1 million a day, rule-based policy (calculation)",
              "value": "0.000016 (0.000027 at the 95th percentile time)",
              "n": 20000,
              "note": "Decisions per second × assumed median or p95 time. These are scenarios, not actual mean concurrency or confidence bounds."
            },
            {
              "quantity": "Decisions per second at 10 million a day (calculation)",
              "value": "115.7 per second",
              "n": null,
              "note": "A steady rate over 24 hours."
            },
            {
              "quantity": "Daily cost at 10 million decisions, rule-based policy (calculation)",
              "value": "$0",
              "n": 20000,
              "note": "$0 a year at a flat volume."
            },
            {
              "quantity": "Daily cost at 10 million decisions, Jev (calculation)",
              "value": "$337.00",
              "n": 82,
              "note": "$123,005 a year at a flat volume."
            },
            {
              "quantity": "Daily cost at 10 million decisions, Sonnet 5.5 router (calculation)",
              "value": "$73,240",
              "n": 82,
              "note": "$26,732,600 a year at a flat volume."
            },
            {
              "quantity": "Daily cost at 10 million decisions, Haiku 4.5 router (calculation)",
              "value": "$89,240",
              "n": 82,
              "note": "$32,572,600 a year at a flat volume."
            },
            {
              "quantity": "Decisions in flight at 10 million a day, rule-based policy (calculation)",
              "value": "0.00016 (0.00027 at the 95th percentile time)",
              "n": 20000,
              "note": "Decisions per second × assumed median or p95 time. These are scenarios, not actual mean concurrency or confidence bounds."
            },
            {
              "quantity": "Decisions in flight at 10 million a day, Jev (calculation)",
              "value": "15.8 (22.7 at the 95th percentile time)",
              "n": 246,
              "note": "Decisions per second × assumed median or p95 time. These are scenarios, not actual mean concurrency or confidence bounds."
            },
            {
              "quantity": "Decisions in flight at 10 million a day, Sonnet 5.5 router (calculation)",
              "value": "300.7 (497.5 at the 95th percentile time)",
              "n": 82,
              "note": "Decisions per second × assumed median or p95 time. These are scenarios, not actual mean concurrency or confidence bounds."
            },
            {
              "quantity": "Decisions in flight at 10 million a day, Haiku 4.5 router (calculation)",
              "value": "1,467 (3,983 at the 95th percentile time)",
              "n": 82,
              "note": "Decisions per second × assumed median or p95 time. These are scenarios, not actual mean concurrency or confidence bounds."
            },
            {
              "quantity": "Waiting hours a day at 1 million decisions, Haiku 4.5 router (calculation)",
              "value": "3,521 hours (9,559 at the 95th percentile time)",
              "n": 82,
              "note": "Decisions × assumed median or p95 time. Actual total waiting requires the mean; these are not confidence bounds."
            },
            {
              "quantity": "Work cost of 1 million model calls a day at the recorded cost per call (calculation)",
              "value": "$61,131",
              "n": 48,
              "note": "A notional list-price cost; your tasks may differ."
            },
            {
              "quantity": "Router cost as a share of the work cost per call, Sonnet 5.5 router (calculation)",
              "value": "12.0%",
              "n": 82,
              "note": "Cost per decision ÷ recorded work cost per model call, if the router decides every call."
            },
            {
              "quantity": "Router cost as a share of the work cost per call, Haiku 4.5 router (calculation)",
              "value": "14.6%",
              "n": 82,
              "note": "Cost per decision ÷ recorded work cost per model call, if the router decides every call."
            },
            {
              "quantity": "Router cost as a share of the work cost per call, Jev (calculation)",
              "value": "0.06%",
              "n": 82,
              "note": "Cost per decision ÷ recorded work cost per model call, if the router decides every call."
            },
            {
              "quantity": "Rule-based policy: measured decisions per second in one batch",
              "value": "469,409 per second",
              "n": 200000,
              "note": "One process, one 200,000-iteration throughput batch of the routing-overhead study. Database reads and the decision-record write are not included."
            },
            {
              "quantity": "Share of that measured policy rate used at 10 million decisions a day (calculation)",
              "value": "0.025%",
              "n": null,
              "note": "Decisions per second at that volume ÷ measured decisions per second."
            }
          ]
        },
        {
          "id": "routing-at-scale-grid",
          "title": "Every scenario, volume and router (calculation)",
          "columns": [
            {
              "key": "scenario",
              "label": "Scenario",
              "unit": "text"
            },
            {
              "key": "calls",
              "label": "Model calls a day",
              "unit": "count"
            },
            {
              "key": "router",
              "label": "Router",
              "unit": "text"
            },
            {
              "key": "decisions",
              "label": "Decisions routed a day",
              "unit": "count"
            },
            {
              "key": "perSecond",
              "label": "Decisions per second (24-hour average)",
              "unit": "count"
            },
            {
              "key": "costDay",
              "label": "Cost per day",
              "unit": "usd"
            },
            {
              "key": "costYear",
              "label": "Cost per year (× 365)",
              "unit": "usd"
            },
            {
              "key": "flightP50",
              "label": "In flight at median time",
              "unit": "calls"
            },
            {
              "key": "flightP95",
              "label": "In flight at 95th percentile time",
              "unit": "calls"
            },
            {
              "key": "hoursP50",
              "label": "Waiting hours a day at median time",
              "unit": "count"
            },
            {
              "key": "hoursP95",
              "label": "Waiting hours a day at 95th percentile time",
              "unit": "count"
            }
          ],
          "rows": [
            {
              "scenario": "Every model call routed",
              "calls": 10000,
              "router": "Deterministic routing policy",
              "decisions": 10000,
              "perSecond": 0.11574,
              "costDay": 0,
              "costYear": 0,
              "flightP50": 1.6435e-7,
              "flightP95": 2.6968e-7,
              "hoursP50": 0.0000039444,
              "hoursP95": 0.0000064722
            },
            {
              "scenario": "Every model call routed",
              "calls": 10000,
              "router": "Jev 1.13 (TypeSafe)",
              "decisions": 10000,
              "perSecond": 0.11574,
              "costDay": 0.337,
              "costYear": 123.01,
              "flightP50": 0.015799,
              "flightP95": 0.02265,
              "hoursP50": 0.37917,
              "hoursP95": 0.54361
            },
            {
              "scenario": "Every model call routed",
              "calls": 10000,
              "router": "Claude Sonnet 5.5 (low) · Claude Code",
              "decisions": 10000,
              "perSecond": 0.11574,
              "costDay": 73.24,
              "costYear": 26733,
              "flightP50": 0.30069,
              "flightP95": 0.49745,
              "hoursP50": 7.2167,
              "hoursP95": 11.939
            },
            {
              "scenario": "Every model call routed",
              "calls": 10000,
              "router": "Claude Haiku 4.5 · Claude Code",
              "decisions": 10000,
              "perSecond": 0.11574,
              "costDay": 89.24,
              "costYear": 32573,
              "flightP50": 1.4669,
              "flightP95": 3.983,
              "hoursP50": 35.206,
              "hoursP95": 95.592
            },
            {
              "scenario": "Every model call routed",
              "calls": 100000,
              "router": "Deterministic routing policy",
              "decisions": 100000,
              "perSecond": 1.1574,
              "costDay": 0,
              "costYear": 0,
              "flightP50": 0.0000016435,
              "flightP95": 0.0000026968,
              "hoursP50": 0.000039444,
              "hoursP95": 0.000064722
            },
            {
              "scenario": "Every model call routed",
              "calls": 100000,
              "router": "Jev 1.13 (TypeSafe)",
              "decisions": 100000,
              "perSecond": 1.1574,
              "costDay": 3.37,
              "costYear": 1230,
              "flightP50": 0.15799,
              "flightP95": 0.2265,
              "hoursP50": 3.7917,
              "hoursP95": 5.4361
            },
            {
              "scenario": "Every model call routed",
              "calls": 100000,
              "router": "Claude Sonnet 5.5 (low) · Claude Code",
              "decisions": 100000,
              "perSecond": 1.1574,
              "costDay": 732.4,
              "costYear": 267330,
              "flightP50": 3.0069,
              "flightP95": 4.9745,
              "hoursP50": 72.167,
              "hoursP95": 119.39
            },
            {
              "scenario": "Every model call routed",
              "calls": 100000,
              "router": "Claude Haiku 4.5 · Claude Code",
              "decisions": 100000,
              "perSecond": 1.1574,
              "costDay": 892.4,
              "costYear": 325730,
              "flightP50": 14.669,
              "flightP95": 39.83,
              "hoursP50": 352.06,
              "hoursP95": 955.92
            },
            {
              "scenario": "Every model call routed",
              "calls": 1000000,
              "router": "Deterministic routing policy",
              "decisions": 1000000,
              "perSecond": 11.574,
              "costDay": 0,
              "costYear": 0,
              "flightP50": 0.000016435,
              "flightP95": 0.000026968,
              "hoursP50": 0.00039444,
              "hoursP95": 0.00064722
            },
            {
              "scenario": "Every model call routed",
              "calls": 1000000,
              "router": "Jev 1.13 (TypeSafe)",
              "decisions": 1000000,
              "perSecond": 11.574,
              "costDay": 33.7,
              "costYear": 12301,
              "flightP50": 1.5799,
              "flightP95": 2.265,
              "hoursP50": 37.917,
              "hoursP95": 54.361
            },
            {
              "scenario": "Every model call routed",
              "calls": 1000000,
              "router": "Claude Sonnet 5.5 (low) · Claude Code",
              "decisions": 1000000,
              "perSecond": 11.574,
              "costDay": 7324,
              "costYear": 2673300,
              "flightP50": 30.069,
              "flightP95": 49.745,
              "hoursP50": 721.67,
              "hoursP95": 1193.9
            },
            {
              "scenario": "Every model call routed",
              "calls": 1000000,
              "router": "Claude Haiku 4.5 · Claude Code",
              "decisions": 1000000,
              "perSecond": 11.574,
              "costDay": 8924,
              "costYear": 3257300,
              "flightP50": 146.69,
              "flightP95": 398.3,
              "hoursP50": 3520.6,
              "hoursP95": 9559.2
            },
            {
              "scenario": "Every model call routed",
              "calls": 10000000,
              "router": "Deterministic routing policy",
              "decisions": 10000000,
              "perSecond": 115.74,
              "costDay": 0,
              "costYear": 0,
              "flightP50": 0.00016435,
              "flightP95": 0.00026968,
              "hoursP50": 0.0039444,
              "hoursP95": 0.0064722
            },
            {
              "scenario": "Every model call routed",
              "calls": 10000000,
              "router": "Jev 1.13 (TypeSafe)",
              "decisions": 10000000,
              "perSecond": 115.74,
              "costDay": 337,
              "costYear": 123010,
              "flightP50": 15.799,
              "flightP95": 22.65,
              "hoursP50": 379.17,
              "hoursP95": 543.61
            },
            {
              "scenario": "Every model call routed",
              "calls": 10000000,
              "router": "Claude Sonnet 5.5 (low) · Claude Code",
              "decisions": 10000000,
              "perSecond": 115.74,
              "costDay": 73240,
              "costYear": 26733000,
              "flightP50": 300.69,
              "flightP95": 497.45,
              "hoursP50": 7216.7,
              "hoursP95": 11939
            },
            {
              "scenario": "Every model call routed",
              "calls": 10000000,
              "router": "Claude Haiku 4.5 · Claude Code",
              "decisions": 10000000,
              "perSecond": 115.74,
              "costDay": 89240,
              "costYear": 32573000,
              "flightP50": 1466.9,
              "flightP95": 3983,
              "hoursP50": 35206,
              "hoursP95": 95592
            },
            {
              "scenario": "Only System One decisions routed (14.1% of calls)",
              "calls": 10000,
              "router": "Deterministic routing policy",
              "decisions": 1414,
              "perSecond": 0.016366,
              "costDay": 0,
              "costYear": 0,
              "flightP50": 2.3239e-8,
              "flightP95": 3.8132e-8,
              "hoursP50": 5.5774e-7,
              "hoursP95": 9.1517e-7
            },
            {
              "scenario": "Only System One decisions routed (14.1% of calls)",
              "calls": 10000,
              "router": "Jev 1.13 (TypeSafe)",
              "decisions": 1414,
              "perSecond": 0.016366,
              "costDay": 0.047652,
              "costYear": 17.393,
              "flightP50": 0.0022339,
              "flightP95": 0.0032028,
              "hoursP50": 0.053614,
              "hoursP95": 0.076867
            },
            {
              "scenario": "Only System One decisions routed (14.1% of calls)",
              "calls": 10000,
              "router": "Claude Sonnet 5.5 (low) · Claude Code",
              "decisions": 1414,
              "perSecond": 0.016366,
              "costDay": 10.356,
              "costYear": 3780,
              "flightP50": 0.042518,
              "flightP95": 0.07034,
              "hoursP50": 1.0204,
              "hoursP95": 1.6882
            },
            {
              "scenario": "Only System One decisions routed (14.1% of calls)",
              "calls": 10000,
              "router": "Claude Haiku 4.5 · Claude Code",
              "decisions": 1414,
              "perSecond": 0.016366,
              "costDay": 12.619,
              "costYear": 4605.8,
              "flightP50": 0.20742,
              "flightP95": 0.56319,
              "hoursP50": 4.9781,
              "hoursP95": 13.517
            },
            {
              "scenario": "Only System One decisions routed (14.1% of calls)",
              "calls": 100000,
              "router": "Deterministic routing policy",
              "decisions": 14141,
              "perSecond": 0.16367,
              "costDay": 0,
              "costYear": 0,
              "flightP50": 2.3241e-7,
              "flightP95": 3.8135e-7,
              "hoursP50": 0.0000055778,
              "hoursP95": 0.0000091524
            },
            {
              "scenario": "Only System One decisions routed (14.1% of calls)",
              "calls": 100000,
              "router": "Jev 1.13 (TypeSafe)",
              "decisions": 14141,
              "perSecond": 0.16367,
              "costDay": 0.47655,
              "costYear": 173.94,
              "flightP50": 0.022341,
              "flightP95": 0.03203,
              "hoursP50": 0.53618,
              "hoursP95": 0.76872
            },
            {
              "scenario": "Only System One decisions routed (14.1% of calls)",
              "calls": 100000,
              "router": "Claude Sonnet 5.5 (low) · Claude Code",
              "decisions": 14141,
              "perSecond": 0.16367,
              "costDay": 103.57,
              "costYear": 37803,
              "flightP50": 0.42521,
              "flightP95": 0.70345,
              "hoursP50": 10.205,
              "hoursP95": 16.883
            },
            {
              "scenario": "Only System One decisions routed (14.1% of calls)",
              "calls": 100000,
              "router": "Claude Haiku 4.5 · Claude Code",
              "decisions": 14141,
              "perSecond": 0.16367,
              "costDay": 126.19,
              "costYear": 46061,
              "flightP50": 2.0743,
              "flightP95": 5.6323,
              "hoursP50": 49.784,
              "hoursP95": 135.18
            },
            {
              "scenario": "Only System One decisions routed (14.1% of calls)",
              "calls": 1000000,
              "router": "Deterministic routing policy",
              "decisions": 141414,
              "perSecond": 1.6367,
              "costDay": 0,
              "costYear": 0,
              "flightP50": 0.0000023242,
              "flightP95": 0.0000038136,
              "hoursP50": 0.00005578,
              "hoursP95": 0.000091526
            },
            {
              "scenario": "Only System One decisions routed (14.1% of calls)",
              "calls": 1000000,
              "router": "Jev 1.13 (TypeSafe)",
              "decisions": 141414,
              "perSecond": 1.6367,
              "costDay": 4.7657,
              "costYear": 1739.5,
              "flightP50": 0.22341,
              "flightP95": 0.32031,
              "hoursP50": 5.3619,
              "hoursP95": 7.6874
            },
            {
              "scenario": "Only System One decisions routed (14.1% of calls)",
              "calls": 1000000,
              "router": "Claude Sonnet 5.5 (low) · Claude Code",
              "decisions": 141414,
              "perSecond": 1.6367,
              "costDay": 1035.7,
              "costYear": 378040,
              "flightP50": 4.2522,
              "flightP95": 7.0347,
              "hoursP50": 102.05,
              "hoursP95": 168.83
            },
            {
              "scenario": "Only System One decisions routed (14.1% of calls)",
              "calls": 1000000,
              "router": "Claude Haiku 4.5 · Claude Code",
              "decisions": 141414,
              "perSecond": 1.6367,
              "costDay": 1262,
              "costYear": 460620,
              "flightP50": 20.744,
              "flightP95": 56.325,
              "hoursP50": 497.86,
              "hoursP95": 1351.8
            },
            {
              "scenario": "Only System One decisions routed (14.1% of calls)",
              "calls": 10000000,
              "router": "Deterministic routing policy",
              "decisions": 1414141,
              "perSecond": 16.367,
              "costDay": 0,
              "costYear": 0,
              "flightP50": 0.000023242,
              "flightP95": 0.000038136,
              "hoursP50": 0.0005578,
              "hoursP95": 0.00091526
            },
            {
              "scenario": "Only System One decisions routed (14.1% of calls)",
              "calls": 10000000,
              "router": "Jev 1.13 (TypeSafe)",
              "decisions": 1414141,
              "perSecond": 16.367,
              "costDay": 47.657,
              "costYear": 17395,
              "flightP50": 2.2341,
              "flightP95": 3.2031,
              "hoursP50": 53.62,
              "hoursP95": 76.874
            },
            {
              "scenario": "Only System One decisions routed (14.1% of calls)",
              "calls": 10000000,
              "router": "Claude Sonnet 5.5 (low) · Claude Code",
              "decisions": 1414141,
              "perSecond": 16.367,
              "costDay": 10357,
              "costYear": 3780400,
              "flightP50": 42.522,
              "flightP95": 70.347,
              "hoursP50": 1020.5,
              "hoursP95": 1688.3
            },
            {
              "scenario": "Only System One decisions routed (14.1% of calls)",
              "calls": 10000000,
              "router": "Claude Haiku 4.5 · Claude Code",
              "decisions": 1414141,
              "perSecond": 16.367,
              "costDay": 12620,
              "costYear": 4606200,
              "flightP50": 207.44,
              "flightP95": 563.25,
              "hoursP50": 4978.6,
              "hoursP95": 13518
            }
          ]
        }
      ],
      "related": [
        "routing-overhead",
        "routing-jev-vs-llm"
      ]
    }
  ],
  "entities": [
    {
      "slug": "claude-sonnet-5-5",
      "name": "Claude Sonnet 5.5",
      "vendor": "Anthropic",
      "kind": "model",
      "description": "Anthropic’s mid-tier Claude model. Agent’s default coding model; measured here through Claude Code at low, medium, high and default effort, and as an LLM router.",
      "aliases": [
        "Claude Sonnet 5.5",
        "Sonnet 5.5"
      ],
      "facts": [
        {
          "studySlug": "swe-bench-opus-vs-sonnet",
          "chartId": "swebench-opus-sonnet-resolved",
          "series": "Resolved",
          "point": "Claude Sonnet 5.5 (Agent, older builds)",
          "metric": "swebench-opus-sonnet-resolved",
          "label": "Resolved on the same 3 SWE-bench Verified instances (interim)",
          "value": 0.3333,
          "unit": "rate",
          "display": "33% (1/3)",
          "n": 3,
          "ci": [
            0.0615,
            0.7923
          ],
          "spanKind": "ci95",
          "context": "Agent · older builds · SWE-bench Verified, interim paired probe"
        },
        {
          "studySlug": "swe-bench-opus-vs-sonnet",
          "chartId": "swebench-opus-sonnet-cost-per-attempt",
          "series": "List-price cost per attempt",
          "point": "Claude Sonnet 5.5 (Agent, older builds)",
          "metric": "swebench-opus-sonnet-cost-per-attempt",
          "label": "List-price cost per attempt (calculation)",
          "value": 2.88,
          "unit": "usd",
          "display": "$2.88",
          "n": 3,
          "calculation": true,
          "context": "Agent · older builds · SWE-bench Verified, interim paired probe"
        },
        {
          "studySlug": "swe-bench-opus-vs-sonnet",
          "chartId": "swebench-opus-sonnet-minutes",
          "series": "Worker minutes per attempt",
          "point": "Claude Sonnet 5.5 (Agent, older builds)",
          "metric": "swebench-opus-sonnet-minutes",
          "label": "Worker time per attempt",
          "value": 9.37,
          "unit": "minutes",
          "display": "9.4 min",
          "n": 3,
          "range": [
            4.74,
            15
          ],
          "spanKind": "minmax",
          "context": "Agent · older builds · SWE-bench Verified, interim paired probe"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-pass-rate",
          "series": "Pass rate",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "h2h-pass-rate",
          "label": "Pass rate on five validated tasks",
          "value": 0.8,
          "unit": "rate",
          "display": "80% (12/15)",
          "n": 15,
          "ci": [
            0.5481,
            0.9295
          ],
          "spanKind": "ci95",
          "context": "Claude Code · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-total-latency",
          "series": "Total time per call",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "h2h-total-latency",
          "label": "Total time per call",
          "value": 2.31,
          "unit": "seconds",
          "display": "2.31 s",
          "n": 15,
          "range": [
            2.17,
            7.73
          ],
          "spanKind": "minmax",
          "context": "Claude Code · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-first-useful-latency",
          "series": "Time to first useful output",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "h2h-first-useful-latency",
          "label": "Time to first useful output",
          "value": 1.56,
          "unit": "seconds",
          "display": "1.56 s",
          "n": 15,
          "range": [
            0.99,
            6.39
          ],
          "spanKind": "minmax",
          "context": "Claude Code · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-input-tokens",
          "series": "Cache read",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "h2h-input-tokens/Cache read",
          "label": "Input tokens per call: what the CLI sends (Cache read)",
          "value": 1401,
          "unit": "tokens",
          "display": "1,401",
          "n": 15,
          "context": "Claude Code · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-input-tokens",
          "series": "Other input",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "h2h-input-tokens/Other input",
          "label": "Input tokens per call: what the CLI sends (Other input)",
          "value": 685,
          "unit": "tokens",
          "display": "685",
          "n": 15,
          "context": "Claude Code · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-output-tokens",
          "series": "Output tokens",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "h2h-output-tokens/Output tokens",
          "label": "Output tokens per call (Output tokens)",
          "value": 107,
          "unit": "tokens",
          "display": "107",
          "n": 15,
          "context": "Claude Code · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-list-price-per-call",
          "series": "Cost per call",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "h2h-list-price-per-call",
          "label": "List-price cost per call (calculation)",
          "value": 0.0036,
          "unit": "usd",
          "display": "$0.0036",
          "n": 15,
          "range": [
            0.00342,
            0.01021
          ],
          "spanKind": "minmax",
          "calculation": true,
          "context": "Claude Code · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-cost-per-pass",
          "series": "Cost per pass",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "h2h-cost-per-pass",
          "label": "List-price cost per passing answer (calculation)",
          "value": 0.00624,
          "unit": "usd",
          "display": "$0.0062",
          "n": 15,
          "calculation": true,
          "context": "Claude Code · five short validated tasks"
        },
        {
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-pass-rate",
          "series": "Strict pass",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "hard-h2h-pass-rate/Strict pass",
          "label": "Pass rate on eight hard tasks (Strict pass)",
          "value": 1,
          "unit": "rate",
          "display": "100% (24/24)",
          "n": 24,
          "ci": [
            0.862,
            1
          ],
          "spanKind": "ci95",
          "context": "Claude Code · eight hard validated tasks"
        },
        {
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-pass-rate",
          "series": "Lenient (format misses counted)",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "hard-h2h-pass-rate/Lenient (format misses counted)",
          "label": "Pass rate on eight hard tasks (Lenient (format misses counted))",
          "value": 1,
          "unit": "rate",
          "display": "100% (24/24)",
          "n": 24,
          "ci": [
            0.862,
            1
          ],
          "spanKind": "ci95",
          "context": "Claude Code · eight hard validated tasks"
        },
        {
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-total-latency",
          "series": "Total time per call on hard tasks (separate batches)",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "hard-h2h-total-latency",
          "label": "Total time per call on hard tasks (separate batches)",
          "value": 7.75,
          "unit": "seconds",
          "display": "7.75 s",
          "n": 24,
          "range": [
            2.26,
            34.79
          ],
          "spanKind": "minmax",
          "context": "Claude Code · eight hard validated tasks"
        },
        {
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-first-useful-latency",
          "series": "Time to first useful output on hard tasks",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "hard-h2h-first-useful-latency",
          "label": "Time to first useful output on hard tasks",
          "value": 5.95,
          "unit": "seconds",
          "display": "5.95 s",
          "n": 24,
          "range": [
            0.86,
            30.57
          ],
          "spanKind": "minmax",
          "context": "Claude Code · eight hard validated tasks"
        },
        {
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-output-tokens",
          "series": "Output tokens",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "hard-h2h-output-tokens/Output tokens",
          "label": "Output tokens per call on hard tasks (Output tokens)",
          "value": 1050,
          "unit": "tokens",
          "display": "1,050",
          "n": 24,
          "context": "Claude Code · eight hard validated tasks"
        },
        {
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-cost-per-pass",
          "series": "Cost per strict pass",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "hard-h2h-cost-per-pass",
          "label": "List-price cost per strict pass on hard tasks (calculation)",
          "value": 0.01435,
          "unit": "usd",
          "display": "$0.014",
          "n": 24,
          "calculation": true,
          "context": "Claude Code · eight hard validated tasks"
        },
        {
          "studySlug": "coding-agents-head-to-head",
          "chartId": "coding-agents-pass-rate",
          "series": "Passed every hidden check",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "coding-agents-pass-rate",
          "label": "Coding sessions that passed every hidden check",
          "value": 1,
          "unit": "rate",
          "display": "100% (12/12)",
          "n": 12,
          "ci": [
            0.7575,
            1
          ],
          "spanKind": "ci95",
          "context": "Claude Code · six small repository tasks with hidden tests"
        },
        {
          "studySlug": "coding-agents-head-to-head",
          "chartId": "coding-agents-wall-time",
          "series": "Wall time per session",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "coding-agents-wall-time",
          "label": "Time per coding session",
          "value": 23.1,
          "unit": "seconds",
          "display": "23.1 s",
          "n": 12,
          "range": [
            18.7,
            44.5
          ],
          "spanKind": "minmax",
          "context": "Claude Code · six small repository tasks with hidden tests"
        },
        {
          "studySlug": "coding-agents-head-to-head",
          "chartId": "coding-agents-tool-calls",
          "series": "Tool calls per session",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "coding-agents-tool-calls",
          "label": "Tool calls per coding session",
          "value": 7.5,
          "unit": "calls",
          "display": "7.5",
          "n": 12,
          "range": [
            3,
            14
          ],
          "spanKind": "minmax",
          "context": "Claude Code · six small repository tasks with hidden tests"
        },
        {
          "studySlug": "coding-agents-head-to-head",
          "chartId": "coding-agents-cost-per-pass",
          "series": "List-price cost per pass",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "coding-agents-cost-per-pass",
          "label": "List-price cost per passing coding session (calculation)",
          "value": 0.085,
          "unit": "usd",
          "display": "$0.085",
          "n": 12,
          "calculation": true,
          "context": "Claude Code · six small repository tasks with hidden tests"
        },
        {
          "studySlug": "effort-ladder",
          "chartId": "effort-ladder-pass-rate",
          "series": "Strict pass",
          "point": "Claude Sonnet 5.5 (low) · Claude Code",
          "metric": "effort-ladder-pass-rate",
          "label": "Strict pass rate by effort on eight hard tasks",
          "value": 1,
          "unit": "rate",
          "display": "100% (16/16)",
          "n": 16,
          "ci": [
            0.8064,
            1
          ],
          "spanKind": "ci95",
          "context": "Claude Code · effort low · eight hard validated tasks, effort ladder"
        },
        {
          "studySlug": "effort-ladder",
          "chartId": "effort-ladder-pass-rate",
          "series": "Strict pass",
          "point": "Claude Sonnet 5.5 (medium) · Claude Code",
          "metric": "effort-ladder-pass-rate",
          "label": "Strict pass rate by effort on eight hard tasks",
          "value": 1,
          "unit": "rate",
          "display": "100% (16/16)",
          "n": 16,
          "ci": [
            0.8064,
            1
          ],
          "spanKind": "ci95",
          "context": "Claude Code · effort medium · eight hard validated tasks, effort ladder"
        },
        {
          "studySlug": "effort-ladder",
          "chartId": "effort-ladder-pass-rate",
          "series": "Strict pass",
          "point": "Claude Sonnet 5.5 (high) · Claude Code",
          "metric": "effort-ladder-pass-rate",
          "label": "Strict pass rate by effort on eight hard tasks",
          "value": 1,
          "unit": "rate",
          "display": "100% (16/16)",
          "n": 16,
          "ci": [
            0.8064,
            1
          ],
          "spanKind": "ci95",
          "context": "Claude Code · effort high · eight hard validated tasks, effort ladder"
        },
        {
          "studySlug": "effort-ladder",
          "chartId": "effort-ladder-pass-rate",
          "series": "Strict pass",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "effort-ladder-pass-rate",
          "label": "Strict pass rate by effort on eight hard tasks",
          "value": 1,
          "unit": "rate",
          "display": "100% (16/16)",
          "n": 16,
          "ci": [
            0.8064,
            1
          ],
          "spanKind": "ci95",
          "context": "Claude Code · eight hard validated tasks, effort ladder"
        },
        {
          "studySlug": "effort-ladder",
          "chartId": "effort-ladder-total-latency",
          "series": "Total time per call by effort on hard tasks",
          "point": "Claude Sonnet 5.5 (low) · Claude Code",
          "metric": "effort-ladder-total-latency",
          "label": "Total time per call by effort on hard tasks",
          "value": 5.82,
          "unit": "seconds",
          "display": "5.82 s",
          "n": 16,
          "range": [
            2.78,
            19.96
          ],
          "spanKind": "minmax",
          "context": "Claude Code · effort low · eight hard validated tasks, effort ladder"
        },
        {
          "studySlug": "effort-ladder",
          "chartId": "effort-ladder-total-latency",
          "series": "Total time per call by effort on hard tasks",
          "point": "Claude Sonnet 5.5 (medium) · Claude Code",
          "metric": "effort-ladder-total-latency",
          "label": "Total time per call by effort on hard tasks",
          "value": 7.63,
          "unit": "seconds",
          "display": "7.63 s",
          "n": 16,
          "range": [
            2.71,
            24.01
          ],
          "spanKind": "minmax",
          "context": "Claude Code · effort medium · eight hard validated tasks, effort ladder"
        },
        {
          "studySlug": "effort-ladder",
          "chartId": "effort-ladder-total-latency",
          "series": "Total time per call by effort on hard tasks",
          "point": "Claude Sonnet 5.5 (high) · Claude Code",
          "metric": "effort-ladder-total-latency",
          "label": "Total time per call by effort on hard tasks",
          "value": 8.81,
          "unit": "seconds",
          "display": "8.81 s",
          "n": 16,
          "range": [
            2.93,
            35.81
          ],
          "spanKind": "minmax",
          "context": "Claude Code · effort high · eight hard validated tasks, effort ladder"
        },
        {
          "studySlug": "effort-ladder",
          "chartId": "effort-ladder-total-latency",
          "series": "Total time per call by effort on hard tasks",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "effort-ladder-total-latency",
          "label": "Total time per call by effort on hard tasks",
          "value": 7.97,
          "unit": "seconds",
          "display": "7.97 s",
          "n": 16,
          "range": [
            2.26,
            21.61
          ],
          "spanKind": "minmax",
          "context": "Claude Code · eight hard validated tasks, effort ladder"
        },
        {
          "studySlug": "effort-ladder",
          "chartId": "effort-ladder-output-tokens",
          "series": "Output tokens",
          "point": "Claude Sonnet 5.5 (low) · Claude Code",
          "metric": "effort-ladder-output-tokens/Output tokens",
          "label": "Output tokens per call by effort on hard tasks (Output tokens)",
          "value": 667,
          "unit": "tokens",
          "display": "667",
          "n": 16,
          "context": "Claude Code · effort low · eight hard validated tasks, effort ladder"
        },
        {
          "studySlug": "effort-ladder",
          "chartId": "effort-ladder-output-tokens",
          "series": "Output tokens",
          "point": "Claude Sonnet 5.5 (medium) · Claude Code",
          "metric": "effort-ladder-output-tokens/Output tokens",
          "label": "Output tokens per call by effort on hard tasks (Output tokens)",
          "value": 770,
          "unit": "tokens",
          "display": "770",
          "n": 16,
          "context": "Claude Code · effort medium · eight hard validated tasks, effort ladder"
        },
        {
          "studySlug": "effort-ladder",
          "chartId": "effort-ladder-output-tokens",
          "series": "Output tokens",
          "point": "Claude Sonnet 5.5 (high) · Claude Code",
          "metric": "effort-ladder-output-tokens/Output tokens",
          "label": "Output tokens per call by effort on hard tasks (Output tokens)",
          "value": 1192,
          "unit": "tokens",
          "display": "1,192",
          "n": 16,
          "context": "Claude Code · effort high · eight hard validated tasks, effort ladder"
        },
        {
          "studySlug": "effort-ladder",
          "chartId": "effort-ladder-output-tokens",
          "series": "Output tokens",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "effort-ladder-output-tokens/Output tokens",
          "label": "Output tokens per call by effort on hard tasks (Output tokens)",
          "value": 1054,
          "unit": "tokens",
          "display": "1,054",
          "n": 16,
          "context": "Claude Code · eight hard validated tasks, effort ladder"
        },
        {
          "studySlug": "effort-ladder",
          "chartId": "effort-ladder-cost-per-pass",
          "series": "Cost per strict pass",
          "point": "Claude Sonnet 5.5 (low) · Claude Code",
          "metric": "effort-ladder-cost-per-pass",
          "label": "List-price cost per strict pass by effort (calculation)",
          "value": 0.01219,
          "unit": "usd",
          "display": "$0.012",
          "n": 16,
          "calculation": true,
          "context": "Claude Code · effort low · eight hard validated tasks, effort ladder"
        },
        {
          "studySlug": "effort-ladder",
          "chartId": "effort-ladder-cost-per-pass",
          "series": "Cost per strict pass",
          "point": "Claude Sonnet 5.5 (medium) · Claude Code",
          "metric": "effort-ladder-cost-per-pass",
          "label": "List-price cost per strict pass by effort (calculation)",
          "value": 0.01352,
          "unit": "usd",
          "display": "$0.014",
          "n": 16,
          "calculation": true,
          "context": "Claude Code · effort medium · eight hard validated tasks, effort ladder"
        },
        {
          "studySlug": "effort-ladder",
          "chartId": "effort-ladder-cost-per-pass",
          "series": "Cost per strict pass",
          "point": "Claude Sonnet 5.5 (high) · Claude Code",
          "metric": "effort-ladder-cost-per-pass",
          "label": "List-price cost per strict pass by effort (calculation)",
          "value": 0.01671,
          "unit": "usd",
          "display": "$0.017",
          "n": 16,
          "calculation": true,
          "context": "Claude Code · effort high · eight hard validated tasks, effort ladder"
        },
        {
          "studySlug": "effort-ladder",
          "chartId": "effort-ladder-cost-per-pass",
          "series": "Cost per strict pass",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "effort-ladder-cost-per-pass",
          "label": "List-price cost per strict pass by effort (calculation)",
          "value": 0.01398,
          "unit": "usd",
          "display": "$0.014",
          "n": 16,
          "calculation": true,
          "context": "Claude Code · eight hard validated tasks, effort ladder"
        },
        {
          "studySlug": "caching-consistency",
          "chartId": "caching-cost-with-without",
          "series": "With the cache, as recorded",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "caching-cost-with-without/With the cache, as recorded",
          "label": "List-price cost of 5-question sessions with and without the cache (calculation) (With the cache, as recorded)",
          "value": 0.135003,
          "unit": "usd",
          "display": "$0.14",
          "n": 15,
          "calculation": true,
          "context": "Claude Code · calculation: 5-turn cached sessions over a fixed ledger"
        },
        {
          "studySlug": "caching-consistency",
          "chartId": "caching-cost-with-without",
          "series": "Without a cache: every input token at the input price",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "caching-cost-with-without/Without a cache: every input token at the input price",
          "label": "List-price cost of 5-question sessions with and without the cache (calculation) (Without a cache: every input token at the input price)",
          "value": 0.269788,
          "unit": "usd",
          "display": "$0.27",
          "n": 15,
          "calculation": true,
          "context": "Claude Code · calculation: 5-turn cached sessions over a fixed ledger"
        },
        {
          "studySlug": "caching-consistency",
          "chartId": "caching-latency-first-vs-later",
          "series": "Turn 1 (writes the ledger to the cache)",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "caching-latency-first-vs-later/Turn 1 (writes the ledger to the cache)",
          "label": "Time per turn: first turn vs later turns in a cached session (Turn 1 (writes the ledger to the cache))",
          "value": 1.64,
          "unit": "seconds",
          "display": "1.64 s",
          "n": 3,
          "range": [
            1.58,
            1.79
          ],
          "spanKind": "minmax",
          "context": "Claude Code · 5-turn cached sessions over a fixed ledger"
        },
        {
          "studySlug": "caching-consistency",
          "chartId": "caching-latency-first-vs-later",
          "series": "Turns 2-5 (read the ledger from the cache)",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "caching-latency-first-vs-later/Turns 2-5 (read the ledger from the cache)",
          "label": "Time per turn: first turn vs later turns in a cached session (Turns 2-5 (read the ledger from the cache))",
          "value": 1.61,
          "unit": "seconds",
          "display": "1.61 s",
          "n": 12,
          "range": [
            1.35,
            5.63
          ],
          "spanKind": "minmax",
          "context": "Claude Code · 5-turn cached sessions over a fixed ledger"
        },
        {
          "studySlug": "caching-consistency",
          "chartId": "consistency-pass-rate",
          "series": "Exact number",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "consistency-pass-rate/Exact number",
          "label": "Same prompt, 10 times: strict pass rate (Exact number)",
          "value": 1,
          "unit": "rate",
          "display": "100% (10/10)",
          "n": 10,
          "ci": [
            0.7225,
            1
          ],
          "spanKind": "ci95",
          "context": "Claude Code · same prompt repeated 10 times"
        },
        {
          "studySlug": "caching-consistency",
          "chartId": "consistency-pass-rate",
          "series": "JSON object",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "consistency-pass-rate/JSON object",
          "label": "Same prompt, 10 times: strict pass rate (JSON object)",
          "value": 1,
          "unit": "rate",
          "display": "100% (10/10)",
          "n": 10,
          "ci": [
            0.7225,
            1
          ],
          "spanKind": "ci95",
          "context": "Claude Code · same prompt repeated 10 times"
        },
        {
          "studySlug": "caching-consistency",
          "chartId": "consistency-pass-rate",
          "series": "Code fix",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "consistency-pass-rate/Code fix",
          "label": "Same prompt, 10 times: strict pass rate (Code fix)",
          "value": 1,
          "unit": "rate",
          "display": "100% (10/10)",
          "n": 10,
          "ci": [
            0.7225,
            1
          ],
          "spanKind": "ci95",
          "context": "Claude Code · same prompt repeated 10 times"
        },
        {
          "studySlug": "caching-consistency",
          "chartId": "consistency-distinct-answers",
          "series": "Exact number",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "consistency-distinct-answers/Exact number",
          "label": "Same prompt, 10 times: how many different answers (Exact number)",
          "value": 1,
          "unit": "count",
          "display": "1",
          "n": 10,
          "context": "Claude Code · same prompt repeated 10 times"
        },
        {
          "studySlug": "caching-consistency",
          "chartId": "consistency-distinct-answers",
          "series": "JSON object",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "consistency-distinct-answers/JSON object",
          "label": "Same prompt, 10 times: how many different answers (JSON object)",
          "value": 1,
          "unit": "count",
          "display": "1",
          "n": 10,
          "context": "Claude Code · same prompt repeated 10 times"
        },
        {
          "studySlug": "caching-consistency",
          "chartId": "consistency-distinct-answers",
          "series": "Code fix",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "consistency-distinct-answers/Code fix",
          "label": "Same prompt, 10 times: how many different answers (Code fix)",
          "value": 3,
          "unit": "count",
          "display": "3",
          "n": 10,
          "context": "Claude Code · same prompt repeated 10 times"
        },
        {
          "studySlug": "caching-consistency",
          "chartId": "consistency-latency-spread",
          "series": "Exact number",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "consistency-latency-spread/Exact number",
          "label": "Same prompt, 10 times: time per call (Exact number)",
          "value": 6.89,
          "unit": "seconds",
          "display": "6.89 s",
          "n": 10,
          "range": [
            5.81,
            7.81
          ],
          "spanKind": "minmax",
          "context": "Claude Code · same prompt repeated 10 times"
        },
        {
          "studySlug": "caching-consistency",
          "chartId": "consistency-latency-spread",
          "series": "JSON object",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "consistency-latency-spread/JSON object",
          "label": "Same prompt, 10 times: time per call (JSON object)",
          "value": 2.89,
          "unit": "seconds",
          "display": "2.89 s",
          "n": 10,
          "range": [
            2.68,
            5.3
          ],
          "spanKind": "minmax",
          "context": "Claude Code · same prompt repeated 10 times"
        },
        {
          "studySlug": "caching-consistency",
          "chartId": "consistency-latency-spread",
          "series": "Code fix",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "consistency-latency-spread/Code fix",
          "label": "Same prompt, 10 times: time per call (Code fix)",
          "value": 2.67,
          "unit": "seconds",
          "display": "2.67 s",
          "n": 10,
          "range": [
            2.32,
            4.34
          ],
          "spanKind": "minmax",
          "context": "Claude Code · same prompt repeated 10 times"
        },
        {
          "studySlug": "agent-memory",
          "chartId": "memory-full-pass",
          "series": "Claude Sonnet 5.5",
          "point": "No memory",
          "metric": "memory-full-pass/No memory",
          "label": "Full pass rate by kind of memory: No memory",
          "value": 0.6,
          "unit": "rate",
          "display": "60% (9/15)",
          "n": 15,
          "ci": [
            0.3575,
            0.8018
          ],
          "spanKind": "ci95",
          "context": ""
        },
        {
          "studySlug": "agent-memory",
          "chartId": "memory-full-pass",
          "series": "Claude Sonnet 5.5",
          "point": "/init CLAUDE.md",
          "metric": "memory-full-pass//init CLAUDE.md",
          "label": "Full pass rate by kind of memory: /init CLAUDE.md",
          "value": 0.6,
          "unit": "rate",
          "display": "60% (9/15)",
          "n": 15,
          "ci": [
            0.3575,
            0.8018
          ],
          "spanKind": "ci95",
          "context": ""
        },
        {
          "studySlug": "agent-memory",
          "chartId": "memory-full-pass",
          "series": "Claude Sonnet 5.5",
          "point": "Curated, 11 lines",
          "metric": "memory-full-pass/Curated, 11 lines",
          "label": "Full pass rate by kind of memory: Curated, 11 lines",
          "value": 1,
          "unit": "rate",
          "display": "100% (15/15)",
          "n": 15,
          "ci": [
            0.7961,
            1
          ],
          "spanKind": "ci95",
          "context": ""
        },
        {
          "studySlug": "agent-memory",
          "chartId": "memory-full-pass",
          "series": "Claude Sonnet 5.5",
          "point": "Raw notes, 60 lines",
          "metric": "memory-full-pass/Raw notes, 60 lines",
          "label": "Full pass rate by kind of memory: Raw notes, 60 lines",
          "value": 0.9333,
          "unit": "rate",
          "display": "93% (14/15)",
          "n": 15,
          "ci": [
            0.7018,
            0.9881
          ],
          "spanKind": "ci95",
          "context": ""
        },
        {
          "studySlug": "agent-memory",
          "chartId": "memory-full-pass",
          "series": "Claude Sonnet 5.5",
          "point": "Dreamed notes",
          "metric": "memory-full-pass/Dreamed notes",
          "label": "Full pass rate by kind of memory: Dreamed notes",
          "value": 1,
          "unit": "rate",
          "display": "100% (15/15)",
          "n": 15,
          "ci": [
            0.7961,
            1
          ],
          "spanKind": "ci95",
          "context": ""
        },
        {
          "studySlug": "agent-memory",
          "chartId": "memory-full-pass",
          "series": "Claude Sonnet 5.5",
          "point": "Handbook, 210 lines",
          "metric": "memory-full-pass/Handbook, 210 lines",
          "label": "Full pass rate by kind of memory: Handbook, 210 lines",
          "value": 1,
          "unit": "rate",
          "display": "100% (15/15)",
          "n": 15,
          "ci": [
            0.7961,
            1
          ],
          "spanKind": "ci95",
          "context": ""
        },
        {
          "studySlug": "agent-memory",
          "chartId": "memory-full-pass",
          "series": "Claude Sonnet 5.5",
          "point": "Stop hook only",
          "metric": "memory-full-pass/Stop hook only",
          "label": "Full pass rate by kind of memory: Stop hook only",
          "value": 0.8,
          "unit": "rate",
          "display": "80% (12/15)",
          "n": 15,
          "ci": [
            0.5481,
            0.9295
          ],
          "spanKind": "ci95",
          "context": ""
        },
        {
          "studySlug": "agent-memory",
          "chartId": "memory-full-pass",
          "series": "Claude Sonnet 5.5",
          "point": "Curated + hook",
          "metric": "memory-full-pass/Curated + hook",
          "label": "Full pass rate by kind of memory: Curated + hook",
          "value": 1,
          "unit": "rate",
          "display": "100% (15/15)",
          "n": 15,
          "ci": [
            0.7961,
            1
          ],
          "spanKind": "ci95",
          "context": ""
        },
        {
          "studySlug": "agent-memory",
          "chartId": "memory-team-knowledge-by-model",
          "series": "Claude Sonnet 5.5",
          "point": "No memory",
          "metric": "memory-team-knowledge-by-model/No memory",
          "label": "Team knowledge followed, Sonnet vs Haiku: No memory",
          "value": 0.4,
          "unit": "rate",
          "display": "40% (6/15)",
          "n": 15,
          "ci": [
            0.1982,
            0.6425
          ],
          "spanKind": "ci95",
          "context": ""
        },
        {
          "studySlug": "agent-memory",
          "chartId": "memory-team-knowledge-by-model",
          "series": "Claude Sonnet 5.5",
          "point": "/init CLAUDE.md",
          "metric": "memory-team-knowledge-by-model//init CLAUDE.md",
          "label": "Team knowledge followed, Sonnet vs Haiku: /init CLAUDE.md",
          "value": 0.6667,
          "unit": "rate",
          "display": "67% (10/15)",
          "n": 15,
          "ci": [
            0.4171,
            0.8482
          ],
          "spanKind": "ci95",
          "context": ""
        },
        {
          "studySlug": "agent-memory",
          "chartId": "memory-team-knowledge-by-model",
          "series": "Claude Sonnet 5.5",
          "point": "Curated, 11 lines",
          "metric": "memory-team-knowledge-by-model/Curated, 11 lines",
          "label": "Team knowledge followed, Sonnet vs Haiku: Curated, 11 lines",
          "value": 1,
          "unit": "rate",
          "display": "100% (15/15)",
          "n": 15,
          "ci": [
            0.7961,
            1
          ],
          "spanKind": "ci95",
          "context": ""
        },
        {
          "studySlug": "agent-memory",
          "chartId": "memory-team-knowledge-by-model",
          "series": "Claude Sonnet 5.5",
          "point": "Raw notes, 60 lines",
          "metric": "memory-team-knowledge-by-model/Raw notes, 60 lines",
          "label": "Team knowledge followed, Sonnet vs Haiku: Raw notes, 60 lines",
          "value": 1,
          "unit": "rate",
          "display": "100% (15/15)",
          "n": 15,
          "ci": [
            0.7961,
            1
          ],
          "spanKind": "ci95",
          "context": ""
        },
        {
          "studySlug": "agent-memory",
          "chartId": "memory-team-knowledge-by-model",
          "series": "Claude Sonnet 5.5",
          "point": "Dreamed notes",
          "metric": "memory-team-knowledge-by-model/Dreamed notes",
          "label": "Team knowledge followed, Sonnet vs Haiku: Dreamed notes",
          "value": 1,
          "unit": "rate",
          "display": "100% (15/15)",
          "n": 15,
          "ci": [
            0.7961,
            1
          ],
          "spanKind": "ci95",
          "context": ""
        },
        {
          "studySlug": "agent-memory",
          "chartId": "memory-team-knowledge-by-model",
          "series": "Claude Sonnet 5.5",
          "point": "Handbook, 210 lines",
          "metric": "memory-team-knowledge-by-model/Handbook, 210 lines",
          "label": "Team knowledge followed, Sonnet vs Haiku: Handbook, 210 lines",
          "value": 1,
          "unit": "rate",
          "display": "100% (15/15)",
          "n": 15,
          "ci": [
            0.7961,
            1
          ],
          "spanKind": "ci95",
          "context": ""
        },
        {
          "studySlug": "agent-memory",
          "chartId": "memory-team-knowledge-by-model",
          "series": "Claude Sonnet 5.5",
          "point": "Stop hook only",
          "metric": "memory-team-knowledge-by-model/Stop hook only",
          "label": "Team knowledge followed, Sonnet vs Haiku: Stop hook only",
          "value": 0.6667,
          "unit": "rate",
          "display": "67% (10/15)",
          "n": 15,
          "ci": [
            0.4171,
            0.8482
          ],
          "spanKind": "ci95",
          "context": ""
        },
        {
          "studySlug": "agent-memory",
          "chartId": "memory-team-knowledge-by-model",
          "series": "Claude Sonnet 5.5",
          "point": "Curated + hook",
          "metric": "memory-team-knowledge-by-model/Curated + hook",
          "label": "Team knowledge followed, Sonnet vs Haiku: Curated + hook",
          "value": 1,
          "unit": "rate",
          "display": "100% (15/15)",
          "n": 15,
          "ci": [
            0.7961,
            1
          ],
          "spanKind": "ci95",
          "context": ""
        },
        {
          "studySlug": "agent-memory",
          "chartId": "memory-broken-test-command",
          "series": "Claude Sonnet 5.5",
          "point": "No memory",
          "metric": "memory-broken-test-command/No memory",
          "label": "A stale README command: who still ran it?: No memory",
          "value": 0.8,
          "unit": "rate",
          "display": "80% (12/15)",
          "n": 15,
          "ci": [
            0.5481,
            0.9295
          ],
          "spanKind": "ci95",
          "polarity": "lower",
          "context": ""
        },
        {
          "studySlug": "agent-memory",
          "chartId": "memory-broken-test-command",
          "series": "Claude Sonnet 5.5",
          "point": "/init CLAUDE.md",
          "metric": "memory-broken-test-command//init CLAUDE.md",
          "label": "A stale README command: who still ran it?: /init CLAUDE.md",
          "value": 0.8667,
          "unit": "rate",
          "display": "87% (13/15)",
          "n": 15,
          "ci": [
            0.6212,
            0.9626
          ],
          "spanKind": "ci95",
          "polarity": "lower",
          "context": ""
        },
        {
          "studySlug": "agent-memory",
          "chartId": "memory-broken-test-command",
          "series": "Claude Sonnet 5.5",
          "point": "Curated, 11 lines",
          "metric": "memory-broken-test-command/Curated, 11 lines",
          "label": "A stale README command: who still ran it?: Curated, 11 lines",
          "value": 0,
          "unit": "rate",
          "display": "0% (0/15)",
          "n": 15,
          "ci": [
            0,
            0.2039
          ],
          "spanKind": "ci95",
          "polarity": "lower",
          "context": ""
        },
        {
          "studySlug": "agent-memory",
          "chartId": "memory-broken-test-command",
          "series": "Claude Sonnet 5.5",
          "point": "Raw notes, 60 lines",
          "metric": "memory-broken-test-command/Raw notes, 60 lines",
          "label": "A stale README command: who still ran it?: Raw notes, 60 lines",
          "value": 0,
          "unit": "rate",
          "display": "0% (0/15)",
          "n": 15,
          "ci": [
            0,
            0.2039
          ],
          "spanKind": "ci95",
          "polarity": "lower",
          "context": ""
        },
        {
          "studySlug": "agent-memory",
          "chartId": "memory-broken-test-command",
          "series": "Claude Sonnet 5.5",
          "point": "Dreamed notes",
          "metric": "memory-broken-test-command/Dreamed notes",
          "label": "A stale README command: who still ran it?: Dreamed notes",
          "value": 0,
          "unit": "rate",
          "display": "0% (0/15)",
          "n": 15,
          "ci": [
            0,
            0.2039
          ],
          "spanKind": "ci95",
          "polarity": "lower",
          "context": ""
        },
        {
          "studySlug": "agent-memory",
          "chartId": "memory-broken-test-command",
          "series": "Claude Sonnet 5.5",
          "point": "Handbook, 210 lines",
          "metric": "memory-broken-test-command/Handbook, 210 lines",
          "label": "A stale README command: who still ran it?: Handbook, 210 lines",
          "value": 0,
          "unit": "rate",
          "display": "0% (0/15)",
          "n": 15,
          "ci": [
            0,
            0.2039
          ],
          "spanKind": "ci95",
          "polarity": "lower",
          "context": ""
        },
        {
          "studySlug": "agent-memory",
          "chartId": "memory-broken-test-command",
          "series": "Claude Sonnet 5.5",
          "point": "Stop hook only",
          "metric": "memory-broken-test-command/Stop hook only",
          "label": "A stale README command: who still ran it?: Stop hook only",
          "value": 0.6,
          "unit": "rate",
          "display": "60% (9/15)",
          "n": 15,
          "ci": [
            0.3575,
            0.8018
          ],
          "spanKind": "ci95",
          "polarity": "lower",
          "context": ""
        },
        {
          "studySlug": "agent-memory",
          "chartId": "memory-broken-test-command",
          "series": "Claude Sonnet 5.5",
          "point": "Curated + hook",
          "metric": "memory-broken-test-command/Curated + hook",
          "label": "A stale README command: who still ran it?: Curated + hook",
          "value": 0,
          "unit": "rate",
          "display": "0% (0/15)",
          "n": 15,
          "ci": [
            0,
            0.2039
          ],
          "spanKind": "ci95",
          "polarity": "lower",
          "context": ""
        },
        {
          "studySlug": "agent-memory",
          "chartId": "memory-cost-per-full-pass",
          "series": "Claude Sonnet 5.5",
          "point": "No memory",
          "metric": "memory-cost-per-full-pass/No memory",
          "label": "List-price cost per fully correct result (calculation): No memory",
          "value": 0.1386,
          "unit": "usd",
          "display": "$0.14",
          "n": 9,
          "calculation": true,
          "context": ""
        },
        {
          "studySlug": "agent-memory",
          "chartId": "memory-cost-per-full-pass",
          "series": "Claude Sonnet 5.5",
          "point": "/init CLAUDE.md",
          "metric": "memory-cost-per-full-pass//init CLAUDE.md",
          "label": "List-price cost per fully correct result (calculation): /init CLAUDE.md",
          "value": 0.1359,
          "unit": "usd",
          "display": "$0.14",
          "n": 9,
          "calculation": true,
          "context": ""
        },
        {
          "studySlug": "agent-memory",
          "chartId": "memory-cost-per-full-pass",
          "series": "Claude Sonnet 5.5",
          "point": "Curated, 11 lines",
          "metric": "memory-cost-per-full-pass/Curated, 11 lines",
          "label": "List-price cost per fully correct result (calculation): Curated, 11 lines",
          "value": 0.0818,
          "unit": "usd",
          "display": "$0.082",
          "n": 15,
          "calculation": true,
          "context": ""
        },
        {
          "studySlug": "agent-memory",
          "chartId": "memory-cost-per-full-pass",
          "series": "Claude Sonnet 5.5",
          "point": "Raw notes, 60 lines",
          "metric": "memory-cost-per-full-pass/Raw notes, 60 lines",
          "label": "List-price cost per fully correct result (calculation): Raw notes, 60 lines",
          "value": 0.1009,
          "unit": "usd",
          "display": "$0.10",
          "n": 14,
          "calculation": true,
          "context": ""
        },
        {
          "studySlug": "agent-memory",
          "chartId": "memory-cost-per-full-pass",
          "series": "Claude Sonnet 5.5",
          "point": "Dreamed notes",
          "metric": "memory-cost-per-full-pass/Dreamed notes",
          "label": "List-price cost per fully correct result (calculation): Dreamed notes",
          "value": 0.0896,
          "unit": "usd",
          "display": "$0.090",
          "n": 15,
          "calculation": true,
          "context": ""
        },
        {
          "studySlug": "agent-memory",
          "chartId": "memory-cost-per-full-pass",
          "series": "Claude Sonnet 5.5",
          "point": "Handbook, 210 lines",
          "metric": "memory-cost-per-full-pass/Handbook, 210 lines",
          "label": "List-price cost per fully correct result (calculation): Handbook, 210 lines",
          "value": 0.1006,
          "unit": "usd",
          "display": "$0.10",
          "n": 15,
          "calculation": true,
          "context": ""
        },
        {
          "studySlug": "agent-memory",
          "chartId": "memory-cost-per-full-pass",
          "series": "Claude Sonnet 5.5",
          "point": "Stop hook only",
          "metric": "memory-cost-per-full-pass/Stop hook only",
          "label": "List-price cost per fully correct result (calculation): Stop hook only",
          "value": 0.1278,
          "unit": "usd",
          "display": "$0.13",
          "n": 12,
          "calculation": true,
          "context": ""
        },
        {
          "studySlug": "agent-memory",
          "chartId": "memory-cost-per-full-pass",
          "series": "Claude Sonnet 5.5",
          "point": "Curated + hook",
          "metric": "memory-cost-per-full-pass/Curated + hook",
          "label": "List-price cost per fully correct result (calculation): Curated + hook",
          "value": 0.0843,
          "unit": "usd",
          "display": "$0.084",
          "n": 15,
          "calculation": true,
          "context": ""
        },
        {
          "studySlug": "agent-memory",
          "chartId": "memory-wall-time",
          "series": "Claude Sonnet 5.5",
          "point": "No memory",
          "metric": "memory-wall-time/No memory",
          "label": "Time per session: No memory",
          "value": 18,
          "unit": "seconds",
          "display": "18.0 s",
          "n": 15,
          "context": ""
        },
        {
          "studySlug": "agent-memory",
          "chartId": "memory-wall-time",
          "series": "Claude Sonnet 5.5",
          "point": "/init CLAUDE.md",
          "metric": "memory-wall-time//init CLAUDE.md",
          "label": "Time per session: /init CLAUDE.md",
          "value": 19,
          "unit": "seconds",
          "display": "19.0 s",
          "n": 15,
          "context": ""
        },
        {
          "studySlug": "agent-memory",
          "chartId": "memory-wall-time",
          "series": "Claude Sonnet 5.5",
          "point": "Curated, 11 lines",
          "metric": "memory-wall-time/Curated, 11 lines",
          "label": "Time per session: Curated, 11 lines",
          "value": 21.9,
          "unit": "seconds",
          "display": "21.9 s",
          "n": 15,
          "context": ""
        },
        {
          "studySlug": "agent-memory",
          "chartId": "memory-wall-time",
          "series": "Claude Sonnet 5.5",
          "point": "Raw notes, 60 lines",
          "metric": "memory-wall-time/Raw notes, 60 lines",
          "label": "Time per session: Raw notes, 60 lines",
          "value": 26.8,
          "unit": "seconds",
          "display": "26.8 s",
          "n": 15,
          "context": ""
        },
        {
          "studySlug": "agent-memory",
          "chartId": "memory-wall-time",
          "series": "Claude Sonnet 5.5",
          "point": "Dreamed notes",
          "metric": "memory-wall-time/Dreamed notes",
          "label": "Time per session: Dreamed notes",
          "value": 27.2,
          "unit": "seconds",
          "display": "27.2 s",
          "n": 15,
          "context": ""
        },
        {
          "studySlug": "agent-memory",
          "chartId": "memory-wall-time",
          "series": "Claude Sonnet 5.5",
          "point": "Handbook, 210 lines",
          "metric": "memory-wall-time/Handbook, 210 lines",
          "label": "Time per session: Handbook, 210 lines",
          "value": 23.6,
          "unit": "seconds",
          "display": "23.6 s",
          "n": 15,
          "context": ""
        },
        {
          "studySlug": "agent-memory",
          "chartId": "memory-wall-time",
          "series": "Claude Sonnet 5.5",
          "point": "Stop hook only",
          "metric": "memory-wall-time/Stop hook only",
          "label": "Time per session: Stop hook only",
          "value": 27.3,
          "unit": "seconds",
          "display": "27.3 s",
          "n": 15,
          "context": ""
        },
        {
          "studySlug": "agent-memory",
          "chartId": "memory-wall-time",
          "series": "Claude Sonnet 5.5",
          "point": "Curated + hook",
          "metric": "memory-wall-time/Curated + hook",
          "label": "Time per session: Curated + hook",
          "value": 22,
          "unit": "seconds",
          "display": "22.0 s",
          "n": 15,
          "context": ""
        },
        {
          "studySlug": "routing-jev-vs-llm",
          "chartId": "routing-exact-decisions",
          "series": "Exact rate",
          "point": "Claude Sonnet 5.5",
          "metric": "routing-exact-decisions",
          "label": "Typed routing decisions answered exactly right",
          "value": 0.939,
          "unit": "rate",
          "display": "94% (77/82)",
          "n": 82,
          "ci": [
            0.8651,
            0.9737
          ],
          "spanKind": "ci95",
          "context": "typed routing decisions · via Claude Code"
        },
        {
          "studySlug": "routing-jev-vs-llm",
          "chartId": "routing-key-accuracy",
          "series": "Key accuracy",
          "point": "Claude Sonnet 5.5",
          "metric": "routing-key-accuracy",
          "label": "Per-question accuracy",
          "value": 0.9742,
          "unit": "rate",
          "display": "97% (189/194)",
          "n": 194,
          "ci": [
            0.9411,
            0.9889
          ],
          "spanKind": "ci95",
          "context": "typed routing decisions · via Claude Code"
        },
        {
          "studySlug": "routing-jev-vs-llm",
          "chartId": "routing-exact-by-decision",
          "series": "Claude Sonnet 5.5",
          "point": "Failure class",
          "metric": "routing-exact-by-decision/Failure class",
          "label": "Exact rate by decision type: Failure class",
          "value": 1,
          "unit": "rate",
          "display": "100% (18/18)",
          "n": 18,
          "ci": [
            0.8241,
            1
          ],
          "spanKind": "ci95",
          "context": "typed routing decisions · via Claude Code"
        },
        {
          "studySlug": "routing-jev-vs-llm",
          "chartId": "routing-exact-by-decision",
          "series": "Claude Sonnet 5.5",
          "point": "Message intent",
          "metric": "routing-exact-by-decision/Message intent",
          "label": "Exact rate by decision type: Message intent",
          "value": 1,
          "unit": "rate",
          "display": "100% (20/20)",
          "n": 20,
          "ci": [
            0.8389,
            1
          ],
          "spanKind": "ci95",
          "context": "typed routing decisions · via Claude Code"
        },
        {
          "studySlug": "routing-jev-vs-llm",
          "chartId": "routing-exact-by-decision",
          "series": "Claude Sonnet 5.5",
          "point": "Is it a rule?",
          "metric": "routing-exact-by-decision/Is it a rule?",
          "label": "Exact rate by decision type: Is it a rule?",
          "value": 1,
          "unit": "rate",
          "display": "100% (12/12)",
          "n": 12,
          "ci": [
            0.7575,
            1
          ],
          "spanKind": "ci95",
          "context": "typed routing decisions · via Claude Code"
        },
        {
          "studySlug": "routing-jev-vs-llm",
          "chartId": "routing-exact-by-decision",
          "series": "Claude Sonnet 5.5",
          "point": "Context shape",
          "metric": "routing-exact-by-decision/Context shape",
          "label": "Exact rate by decision type: Context shape",
          "value": 0.8438,
          "unit": "rate",
          "display": "84% (27/32)",
          "n": 32,
          "ci": [
            0.6825,
            0.9314
          ],
          "spanKind": "ci95",
          "context": "typed routing decisions · via Claude Code"
        },
        {
          "studySlug": "routing-jev-vs-llm",
          "chartId": "routing-cost-per-1000",
          "series": "Cost",
          "point": "Claude Sonnet 5.5",
          "metric": "routing-cost-per-1000",
          "label": "Cost per 1,000 routing decisions",
          "value": 4.996,
          "unit": "usd",
          "display": "$5.00",
          "n": 82,
          "calculation": true,
          "context": "typed routing decisions · via Claude Code"
        },
        {
          "studySlug": "routing-jev-vs-llm",
          "chartId": "routing-decision-latency",
          "series": "Wall time (CLI)",
          "point": "Claude Sonnet 5.5",
          "metric": "routing-decision-latency/Wall time (CLI)",
          "label": "Time per routing decision (Wall time (CLI))",
          "value": 2598,
          "unit": "ms",
          "display": "2,598 ms",
          "n": 82,
          "range": [
            2598,
            4298
          ],
          "spanKind": "p50-p95",
          "context": "typed routing decisions · via Claude Code"
        },
        {
          "studySlug": "routing-jev-vs-llm",
          "chartId": "routing-decision-latency",
          "series": "Model time (API)",
          "point": "Claude Sonnet 5.5",
          "metric": "routing-decision-latency/Model time (API)",
          "label": "Time per routing decision (Model time (API))",
          "value": 1599,
          "unit": "ms",
          "display": "1,599 ms",
          "n": 82,
          "range": [
            1599,
            2574
          ],
          "spanKind": "p50-p95",
          "context": "typed routing decisions · via Claude Code"
        },
        {
          "studySlug": "routing-overhead",
          "chartId": "router-overhead-decision-latency",
          "series": "Decision time",
          "point": "Claude Sonnet 5.5 (effort low, via Claude Code)",
          "metric": "router-overhead-decision-latency",
          "label": "Time to make one routing decision",
          "value": 2597,
          "unit": "ms",
          "display": "2,597 ms",
          "n": 82,
          "range": [
            2597,
            4298
          ],
          "spanKind": "p50-p95",
          "context": "effort low · via Claude Code · routing overhead per decision"
        },
        {
          "studySlug": "routing-overhead",
          "chartId": "router-overhead-cli-vs-model-time",
          "series": "Model API time",
          "point": "Claude Sonnet 5.5 (effort low, via Claude Code)",
          "metric": "router-overhead-cli-vs-model-time/Model API time",
          "label": "Where an LLM router’s time goes: model vs CLI (Model API time)",
          "value": 1596,
          "unit": "ms",
          "display": "1,596 ms",
          "n": 82,
          "range": [
            1596,
            2583
          ],
          "spanKind": "p50-p95",
          "context": "effort low · via Claude Code · routing overhead per decision"
        },
        {
          "studySlug": "routing-overhead",
          "chartId": "router-overhead-cli-vs-model-time",
          "series": "CLI and harness time",
          "point": "Claude Sonnet 5.5 (effort low, via Claude Code)",
          "metric": "router-overhead-cli-vs-model-time/CLI and harness time",
          "label": "Where an LLM router’s time goes: model vs CLI (CLI and harness time)",
          "value": 973,
          "unit": "ms",
          "display": "973 ms",
          "n": 82,
          "range": [
            973,
            1277
          ],
          "spanKind": "p50-p95",
          "context": "effort low · via Claude Code · routing overhead per decision"
        },
        {
          "studySlug": "routing-overhead",
          "chartId": "router-overhead-completed",
          "series": "Completed",
          "point": "Claude Sonnet 5.5 (effort low, via Claude Code)",
          "metric": "router-overhead-completed",
          "label": "Routing calls that returned a decision",
          "value": 1,
          "unit": "rate",
          "display": "100% (82/82)",
          "n": 82,
          "ci": [
            0.9552,
            1
          ],
          "spanKind": "ci95",
          "context": "effort low · via Claude Code · routing overhead per decision"
        },
        {
          "studySlug": "routing-overhead",
          "chartId": "router-overhead-cost-list-price",
          "series": "Cost per 1,000 decisions (list price)",
          "point": "Claude Sonnet 5.5 (effort low, via Claude Code)",
          "metric": "router-overhead-cost-list-price",
          "label": "Cost per 1,000 routing decisions for the model routers (calculation)",
          "value": 4.996,
          "unit": "usd",
          "display": "$5.00",
          "n": 82,
          "calculation": true,
          "context": "effort low · via Claude Code · calculation: Agent’s recorded tokens at this model’s list price · routing overhead per decision"
        },
        {
          "studySlug": "routing-overhead",
          "chartId": "router-overhead-cost-per-1000-tasks",
          "series": "Every model call routed (49.5 per task)",
          "point": "Claude Sonnet 5.5 (effort low, via Claude Code)",
          "metric": "router-overhead-cost-per-1000-tasks/Every model call routed (49.5 per task)",
          "label": "Added routing cost per 1,000 tasks (calculation) (Every model call routed (49.5 per task))",
          "value": 247.3,
          "unit": "usd",
          "display": "$247.30",
          "calculation": true,
          "context": "effort low · via Claude Code · calculation per 1,000 tasks from recorded decision counts"
        },
        {
          "studySlug": "routing-overhead",
          "chartId": "router-overhead-cost-per-1000-tasks",
          "series": "Only System One decisions (7 per task)",
          "point": "Claude Sonnet 5.5 (effort low, via Claude Code)",
          "metric": "router-overhead-cost-per-1000-tasks/Only System One decisions (7 per task)",
          "label": "Added routing cost per 1,000 tasks (calculation) (Only System One decisions (7 per task))",
          "value": 34.97,
          "unit": "usd",
          "display": "$34.97",
          "calculation": true,
          "context": "effort low · via Claude Code · calculation per 1,000 tasks from recorded decision counts"
        },
        {
          "studySlug": "routing-overhead",
          "chartId": "router-overhead-delay-per-task",
          "series": "Every model call routed (49.5 per task)",
          "point": "Claude Sonnet 5.5 (effort low, via Claude Code)",
          "metric": "router-overhead-delay-per-task/Every model call routed (49.5 per task)",
          "label": "Added routing delay per task (calculation) (Every model call routed (49.5 per task))",
          "value": 128.5515,
          "unit": "seconds",
          "display": "128.6 s",
          "calculation": true,
          "context": "effort low · via Claude Code · calculation per task from recorded decision counts, decisions in line"
        },
        {
          "studySlug": "routing-overhead",
          "chartId": "router-overhead-delay-per-task",
          "series": "Only System One decisions (7 per task)",
          "point": "Claude Sonnet 5.5 (effort low, via Claude Code)",
          "metric": "router-overhead-delay-per-task/Only System One decisions (7 per task)",
          "label": "Added routing delay per task (calculation) (Only System One decisions (7 per task))",
          "value": 18.179,
          "unit": "seconds",
          "display": "18.2 s",
          "calculation": true,
          "context": "effort low · via Claude Code · calculation per task from recorded decision counts, decisions in line"
        },
        {
          "studySlug": "cost-thought-experiments",
          "chartId": "repriced-cost-per-resolved",
          "series": "Repriced cost per resolved instance",
          "point": "Claude Sonnet 5.5",
          "metric": "repriced-cost-per-resolved",
          "label": "Thought experiment: the same tokens at other list prices",
          "value": 3.489,
          "unit": "usd",
          "display": "$3.49",
          "calculation": true,
          "context": "calculation: Agent’s recorded tokens at this model’s list price"
        },
        {
          "studySlug": "cost-thought-experiments",
          "chartId": "prompt-cache-savings",
          "series": "With caching (as recorded)",
          "point": "Claude Sonnet 5.5",
          "metric": "prompt-cache-savings/With caching (as recorded)",
          "label": "Thought experiment: what prompt caching saved (With caching (as recorded))",
          "value": 87.23,
          "unit": "usd",
          "display": "$87.23",
          "calculation": true,
          "context": "calculation: Agent’s recorded tokens at this model’s list price"
        },
        {
          "studySlug": "cost-thought-experiments",
          "chartId": "prompt-cache-savings",
          "series": "Without caching",
          "point": "Claude Sonnet 5.5",
          "metric": "prompt-cache-savings/Without caching",
          "label": "Thought experiment: what prompt caching saved (Without caching)",
          "value": 343.33,
          "unit": "usd",
          "display": "$343.33",
          "calculation": true,
          "context": "calculation: Agent’s recorded tokens at this model’s list price"
        },
        {
          "studySlug": "cli-model-latency-tokens",
          "chartId": "scheduler-repair-claude-vs-codex",
          "series": "Total time",
          "point": "Claude Code CLI · Sonnet 5.5 · medium",
          "metric": "scheduler-repair-claude-vs-codex/Total time",
          "label": "Repairing a scheduler: Claude Code vs Codex vs API (Total time)",
          "value": 15,
          "unit": "seconds",
          "display": "15.0 s",
          "n": 3,
          "range": [
            13.89,
            15.89
          ],
          "spanKind": "minmax",
          "context": "Claude Code CLI · effort medium · scheduler repair, 296 checks, 3 runs"
        },
        {
          "studySlug": "cli-model-latency-tokens",
          "chartId": "scheduler-repair-claude-vs-codex",
          "series": "First useful output",
          "point": "Claude Code CLI · Sonnet 5.5 · medium",
          "metric": "scheduler-repair-claude-vs-codex/First useful output",
          "label": "Repairing a scheduler: Claude Code vs Codex vs API (First useful output)",
          "value": 7.55,
          "unit": "seconds",
          "display": "7.55 s",
          "n": 3,
          "range": [
            6.77,
            7.63
          ],
          "spanKind": "minmax",
          "context": "Claude Code CLI · effort medium · scheduler repair, 296 checks, 3 runs"
        },
        {
          "studySlug": "cli-model-latency-tokens",
          "chartId": "scheduler-repair-output-tokens",
          "series": "Output tokens",
          "point": "Claude Code CLI · Sonnet 5.5 · medium",
          "metric": "scheduler-repair-output-tokens/Output tokens",
          "label": "Output tokens to repair the scheduler (Output tokens)",
          "value": 2227,
          "unit": "tokens",
          "display": "2,227",
          "n": 3,
          "context": "Claude Code CLI · effort medium · scheduler repair, 296 checks, 3 runs"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-pass-rate",
          "series": "Strict pass",
          "point": "Claude Sonnet 5.5 (single call) · Claude Code",
          "metric": "agent-loop-pass-rate",
          "label": "Strict pass rate: single call vs agent loop on eight hard tasks",
          "value": 1,
          "unit": "rate",
          "display": "100% (24/24)",
          "n": 24,
          "ci": [
            0.862,
            1
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Code · single call"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-pass-rate",
          "series": "Strict pass",
          "point": "Claude Sonnet 5.5 (agent loop) · Claude Code",
          "metric": "agent-loop-pass-rate",
          "label": "Strict pass rate: single call vs agent loop on eight hard tasks",
          "value": 1,
          "unit": "rate",
          "display": "100% (16/16)",
          "n": 16,
          "ci": [
            0.8064,
            1
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Code · agent loop"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "series": "Claude Sonnet 5.5 (single call) · Claude Code",
          "point": "Interval merge fix",
          "metric": "agent-loop-by-task/Interval merge fix",
          "label": "Strict passes per task: single call vs agent loop: Interval merge fix",
          "value": 1,
          "unit": "rate",
          "display": "100% (3/3)",
          "n": 3,
          "ci": [
            0.4385,
            1
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Code · single call"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "series": "Claude Sonnet 5.5 (single call) · Claude Code",
          "point": "DST day-length fix",
          "metric": "agent-loop-by-task/DST day-length fix",
          "label": "Strict passes per task: single call vs agent loop: DST day-length fix",
          "value": 1,
          "unit": "rate",
          "display": "100% (3/3)",
          "n": 3,
          "ci": [
            0.4385,
            1
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Code · single call"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "series": "Claude Sonnet 5.5 (single call) · Claude Code",
          "point": "CSV parser",
          "metric": "agent-loop-by-task/CSV parser",
          "label": "Strict passes per task: single call vs agent loop: CSV parser",
          "value": 1,
          "unit": "rate",
          "display": "100% (3/3)",
          "n": 3,
          "ci": [
            0.4385,
            1
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Code · single call"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "series": "Claude Sonnet 5.5 (single call) · Claude Code",
          "point": "Event-loop order",
          "metric": "agent-loop-by-task/Event-loop order",
          "label": "Strict passes per task: single call vs agent loop: Event-loop order",
          "value": 1,
          "unit": "rate",
          "display": "100% (3/3)",
          "n": 3,
          "ci": [
            0.4385,
            1
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Code · single call"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "series": "Claude Sonnet 5.5 (single call) · Claude Code",
          "point": "Room schedule",
          "metric": "agent-loop-by-task/Room schedule",
          "label": "Strict passes per task: single call vs agent loop: Room schedule",
          "value": 1,
          "unit": "rate",
          "display": "100% (3/3)",
          "n": 3,
          "ci": [
            0.4385,
            1
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Code · single call"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "series": "Claude Sonnet 5.5 (single call) · Claude Code",
          "point": "SemVer regex",
          "metric": "agent-loop-by-task/SemVer regex",
          "label": "Strict passes per task: single call vs agent loop: SemVer regex",
          "value": 1,
          "unit": "rate",
          "display": "100% (3/3)",
          "n": 3,
          "ci": [
            0.4385,
            1
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Code · single call"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "series": "Claude Sonnet 5.5 (single call) · Claude Code",
          "point": "Money refactor",
          "metric": "agent-loop-by-task/Money refactor",
          "label": "Strict passes per task: single call vs agent loop: Money refactor",
          "value": 1,
          "unit": "rate",
          "display": "100% (3/3)",
          "n": 3,
          "ci": [
            0.4385,
            1
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Code · single call"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "series": "Claude Sonnet 5.5 (single call) · Claude Code",
          "point": "SQL report",
          "metric": "agent-loop-by-task/SQL report",
          "label": "Strict passes per task: single call vs agent loop: SQL report",
          "value": 1,
          "unit": "rate",
          "display": "100% (3/3)",
          "n": 3,
          "ci": [
            0.4385,
            1
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Code · single call"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "series": "Claude Sonnet 5.5 (agent loop) · Claude Code",
          "point": "Interval merge fix",
          "metric": "agent-loop-by-task/Interval merge fix",
          "label": "Strict passes per task: single call vs agent loop: Interval merge fix",
          "value": 1,
          "unit": "rate",
          "display": "100% (2/2)",
          "n": 2,
          "ci": [
            0.3424,
            1
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Code · agent loop"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "series": "Claude Sonnet 5.5 (agent loop) · Claude Code",
          "point": "DST day-length fix",
          "metric": "agent-loop-by-task/DST day-length fix",
          "label": "Strict passes per task: single call vs agent loop: DST day-length fix",
          "value": 1,
          "unit": "rate",
          "display": "100% (2/2)",
          "n": 2,
          "ci": [
            0.3424,
            1
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Code · agent loop"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "series": "Claude Sonnet 5.5 (agent loop) · Claude Code",
          "point": "CSV parser",
          "metric": "agent-loop-by-task/CSV parser",
          "label": "Strict passes per task: single call vs agent loop: CSV parser",
          "value": 1,
          "unit": "rate",
          "display": "100% (2/2)",
          "n": 2,
          "ci": [
            0.3424,
            1
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Code · agent loop"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "series": "Claude Sonnet 5.5 (agent loop) · Claude Code",
          "point": "Event-loop order",
          "metric": "agent-loop-by-task/Event-loop order",
          "label": "Strict passes per task: single call vs agent loop: Event-loop order",
          "value": 1,
          "unit": "rate",
          "display": "100% (2/2)",
          "n": 2,
          "ci": [
            0.3424,
            1
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Code · agent loop"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "series": "Claude Sonnet 5.5 (agent loop) · Claude Code",
          "point": "Room schedule",
          "metric": "agent-loop-by-task/Room schedule",
          "label": "Strict passes per task: single call vs agent loop: Room schedule",
          "value": 1,
          "unit": "rate",
          "display": "100% (2/2)",
          "n": 2,
          "ci": [
            0.3424,
            1
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Code · agent loop"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "series": "Claude Sonnet 5.5 (agent loop) · Claude Code",
          "point": "SemVer regex",
          "metric": "agent-loop-by-task/SemVer regex",
          "label": "Strict passes per task: single call vs agent loop: SemVer regex",
          "value": 1,
          "unit": "rate",
          "display": "100% (2/2)",
          "n": 2,
          "ci": [
            0.3424,
            1
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Code · agent loop"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "series": "Claude Sonnet 5.5 (agent loop) · Claude Code",
          "point": "Money refactor",
          "metric": "agent-loop-by-task/Money refactor",
          "label": "Strict passes per task: single call vs agent loop: Money refactor",
          "value": 1,
          "unit": "rate",
          "display": "100% (2/2)",
          "n": 2,
          "ci": [
            0.3424,
            1
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Code · agent loop"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "series": "Claude Sonnet 5.5 (agent loop) · Claude Code",
          "point": "SQL report",
          "metric": "agent-loop-by-task/SQL report",
          "label": "Strict passes per task: single call vs agent loop: SQL report",
          "value": 1,
          "unit": "rate",
          "display": "100% (2/2)",
          "n": 2,
          "ci": [
            0.3424,
            1
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Code · agent loop"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-total-time",
          "series": "Total time per attempt",
          "point": "Claude Sonnet 5.5 (single call) · Claude Code",
          "metric": "agent-loop-total-time",
          "label": "Total time per attempt: single call vs agent loop",
          "value": 7.75,
          "unit": "seconds",
          "display": "7.75 s",
          "n": 24,
          "range": [
            2.26,
            34.79
          ],
          "spanKind": "minmax",
          "context": "Claude Code · single call"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-total-time",
          "series": "Total time per attempt",
          "point": "Claude Sonnet 5.5 (agent loop) · Claude Code",
          "metric": "agent-loop-total-time",
          "label": "Total time per attempt: single call vs agent loop",
          "value": 7.41,
          "unit": "seconds",
          "display": "7.41 s",
          "n": 16,
          "range": [
            2.75,
            24.19
          ],
          "spanKind": "minmax",
          "context": "Claude Code · agent loop"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-tokens",
          "series": "Input tokens (cache reads included)",
          "point": "Claude Sonnet 5.5 (single call) · Claude Code",
          "metric": "agent-loop-tokens/Input tokens (cache reads included)",
          "label": "Tokens per attempt: single call vs agent loop (Input tokens (cache reads included))",
          "value": 2281,
          "unit": "tokens",
          "display": "2,281",
          "n": 24,
          "range": [
            2234,
            2669
          ],
          "spanKind": "minmax",
          "context": "Claude Code · single call"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-tokens",
          "series": "Input tokens (cache reads included)",
          "point": "Claude Sonnet 5.5 (agent loop) · Claude Code",
          "metric": "agent-loop-tokens/Input tokens (cache reads included)",
          "label": "Tokens per attempt: single call vs agent loop (Input tokens (cache reads included))",
          "value": 9550,
          "unit": "tokens",
          "display": "9,550",
          "n": 16,
          "range": [
            9398,
            33040
          ],
          "spanKind": "minmax",
          "context": "Claude Code · agent loop"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-tokens",
          "series": "Output tokens",
          "point": "Claude Sonnet 5.5 (single call) · Claude Code",
          "metric": "agent-loop-tokens/Output tokens",
          "label": "Tokens per attempt: single call vs agent loop (Output tokens)",
          "value": 1050,
          "unit": "tokens",
          "display": "1,050",
          "n": 24,
          "range": [
            176,
            3895
          ],
          "spanKind": "minmax",
          "context": "Claude Code · single call"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-tokens",
          "series": "Output tokens",
          "point": "Claude Sonnet 5.5 (agent loop) · Claude Code",
          "metric": "agent-loop-tokens/Output tokens",
          "label": "Tokens per attempt: single call vs agent loop (Output tokens)",
          "value": 876,
          "unit": "tokens",
          "display": "876",
          "n": 16,
          "range": [
            219,
            3243
          ],
          "spanKind": "minmax",
          "context": "Claude Code · agent loop"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-tool-calls",
          "series": "Tool calls per attempt",
          "point": "Claude Sonnet 5.5 (agent loop) · Claude Code",
          "metric": "agent-loop-tool-calls",
          "label": "Tool calls per agent-loop attempt",
          "value": 0,
          "unit": "count",
          "display": "0",
          "n": 16,
          "range": [
            0,
            3
          ],
          "spanKind": "minmax",
          "context": "Claude Code · agent loop"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-cost-per-pass",
          "series": "Cost per strict pass",
          "point": "Claude Sonnet 5.5 (single call) · Claude Code",
          "metric": "agent-loop-cost-per-pass",
          "label": "List-price cost per strict pass: single call vs agent loop (calculation)",
          "value": 0.01435,
          "unit": "usd",
          "display": "$0.014",
          "n": 24,
          "calculation": true,
          "context": "Claude Code · single call"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-cost-per-pass",
          "series": "Cost per strict pass",
          "point": "Claude Sonnet 5.5 (agent loop) · Claude Code",
          "metric": "agent-loop-cost-per-pass",
          "label": "List-price cost per strict pass: single call vs agent loop (calculation)",
          "value": 0.02746,
          "unit": "usd",
          "display": "$0.027",
          "n": 16,
          "calculation": true,
          "context": "Claude Code · agent loop"
        },
        {
          "studySlug": "haiku-thinking-on-off",
          "chartId": "haiku-thinking-router-exact",
          "series": "Exact decisions (every scored question right)",
          "point": "Claude Sonnet 5.5 (low) · Claude Code",
          "metric": "haiku-thinking-router-exact/Exact decisions (every scored question right)",
          "label": "Haiku thinking study: typed routing decisions answered exactly right (Exact decisions (every scored question right))",
          "value": 0.939,
          "unit": "rate",
          "display": "94% (77/82)",
          "n": 82,
          "ci": [
            0.8651,
            0.9737
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Code · effort low · typed routing decisions, thinking on vs off"
        },
        {
          "studySlug": "haiku-thinking-on-off",
          "chartId": "haiku-thinking-router-exact",
          "series": "Per-question accuracy",
          "point": "Claude Sonnet 5.5 (low) · Claude Code",
          "metric": "haiku-thinking-router-exact/Per-question accuracy",
          "label": "Haiku thinking study: typed routing decisions answered exactly right (Per-question accuracy)",
          "value": 0.9742,
          "unit": "rate",
          "display": "97% (189/194)",
          "n": 194,
          "ci": [
            0.9411,
            0.9889
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Code · effort low · typed routing decisions, thinking on vs off"
        },
        {
          "studySlug": "haiku-thinking-on-off",
          "chartId": "haiku-thinking-router-latency",
          "series": "Wall time (CLI)",
          "point": "Claude Sonnet 5.5 (low) · Claude Code",
          "metric": "haiku-thinking-router-latency/Wall time (CLI)",
          "label": "Haiku thinking study: time per routing decision (Wall time (CLI))",
          "value": 2.6,
          "unit": "seconds",
          "display": "2.60 s",
          "n": 82,
          "range": [
            2.6,
            4.3
          ],
          "spanKind": "p50-p95",
          "context": "Claude Code · effort low · typed routing decisions, thinking on vs off"
        },
        {
          "studySlug": "haiku-thinking-on-off",
          "chartId": "haiku-thinking-router-latency",
          "series": "Model time (API)",
          "point": "Claude Sonnet 5.5 (low) · Claude Code",
          "metric": "haiku-thinking-router-latency/Model time (API)",
          "label": "Haiku thinking study: time per routing decision (Model time (API))",
          "value": 1.6,
          "unit": "seconds",
          "display": "1.60 s",
          "n": 82,
          "range": [
            1.6,
            2.58
          ],
          "spanKind": "p50-p95",
          "context": "Claude Code · effort low · typed routing decisions, thinking on vs off"
        },
        {
          "studySlug": "haiku-thinking-on-off",
          "chartId": "haiku-thinking-router-tokens",
          "series": "Thinking tokens",
          "point": "Claude Sonnet 5.5 (low) · Claude Code",
          "metric": "haiku-thinking-router-tokens/Thinking tokens",
          "label": "Haiku thinking study: thinking and visible output tokens per routing decision (Thinking tokens)",
          "value": 2,
          "unit": "tokens",
          "display": "2",
          "n": 82,
          "context": "Claude Code · effort low · typed routing decisions, thinking on vs off"
        },
        {
          "studySlug": "haiku-thinking-on-off",
          "chartId": "haiku-thinking-router-tokens",
          "series": "Visible output tokens",
          "point": "Claude Sonnet 5.5 (low) · Claude Code",
          "metric": "haiku-thinking-router-tokens/Visible output tokens",
          "label": "Haiku thinking study: thinking and visible output tokens per routing decision (Visible output tokens)",
          "value": 105,
          "unit": "tokens",
          "display": "105",
          "n": 82,
          "context": "Claude Code · effort low · typed routing decisions, thinking on vs off"
        },
        {
          "studySlug": "haiku-thinking-on-off",
          "chartId": "haiku-thinking-router-cost",
          "series": "Cost per 1,000 decisions",
          "point": "Claude Sonnet 5.5 (low) · Claude Code",
          "metric": "haiku-thinking-router-cost",
          "label": "Haiku thinking study: list-price cost per 1,000 routing decisions (calculation)",
          "value": 7.324,
          "unit": "usd",
          "display": "$7.32",
          "n": 82,
          "calculation": true,
          "context": "Claude Code · effort low · typed routing decisions, thinking on vs off"
        },
        {
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-pass-rate",
          "series": "Strict pass: the whole reply is the right JSON",
          "point": "Claude Sonnet 5.5 (instructions) · Claude Code",
          "metric": "structured-output-pass-rate/Strict pass: the whole reply is the right JSON",
          "label": "Does a JSON schema raise the pass rate? Instructions vs schema mode (Strict pass: the whole reply is the right JSON)",
          "value": 1,
          "unit": "rate",
          "display": "100% (12/12)",
          "n": 12,
          "ci": [
            0.7575,
            1
          ],
          "spanKind": "ci95",
          "calculation": true,
          "polarity": "higher",
          "context": "Claude Code · instructions"
        },
        {
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-pass-rate",
          "series": "Strict pass: the whole reply is the right JSON",
          "point": "Claude Sonnet 5.5 (JSON schema) · Claude Code",
          "metric": "structured-output-pass-rate/Strict pass: the whole reply is the right JSON",
          "label": "Does a JSON schema raise the pass rate? Instructions vs schema mode (Strict pass: the whole reply is the right JSON)",
          "value": 1,
          "unit": "rate",
          "display": "100% (12/12)",
          "n": 12,
          "ci": [
            0.7575,
            1
          ],
          "spanKind": "ci95",
          "calculation": true,
          "polarity": "higher",
          "context": "Claude Code · JSON schema"
        },
        {
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-pass-rate",
          "series": "Right answer in any format (strict pass or format miss)",
          "point": "Claude Sonnet 5.5 (instructions) · Claude Code",
          "metric": "structured-output-pass-rate/Right answer in any format (strict pass or format miss)",
          "label": "Does a JSON schema raise the pass rate? Instructions vs schema mode (Right answer in any format (strict pass or format miss))",
          "value": 1,
          "unit": "rate",
          "display": "100% (12/12)",
          "n": 12,
          "ci": [
            0.7575,
            1
          ],
          "spanKind": "ci95",
          "calculation": true,
          "polarity": "higher",
          "context": "Claude Code · instructions"
        },
        {
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-pass-rate",
          "series": "Right answer in any format (strict pass or format miss)",
          "point": "Claude Sonnet 5.5 (JSON schema) · Claude Code",
          "metric": "structured-output-pass-rate/Right answer in any format (strict pass or format miss)",
          "label": "Does a JSON schema raise the pass rate? Instructions vs schema mode (Right answer in any format (strict pass or format miss))",
          "value": 1,
          "unit": "rate",
          "display": "100% (12/12)",
          "n": 12,
          "ci": [
            0.7575,
            1
          ],
          "spanKind": "ci95",
          "calculation": true,
          "polarity": "higher",
          "context": "Claude Code · JSON schema"
        },
        {
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-outcomes",
          "series": "Strict pass",
          "point": "Claude Sonnet 5.5 (instructions) · Claude Code",
          "metric": "structured-output-outcomes/Strict pass",
          "label": "What each call produced: strict pass, format miss, wrong values or error (Strict pass)",
          "value": 12,
          "unit": "count",
          "display": "12",
          "n": 12,
          "context": "Claude Code · instructions"
        },
        {
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-outcomes",
          "series": "Strict pass",
          "point": "Claude Sonnet 5.5 (JSON schema) · Claude Code",
          "metric": "structured-output-outcomes/Strict pass",
          "label": "What each call produced: strict pass, format miss, wrong values or error (Strict pass)",
          "value": 12,
          "unit": "count",
          "display": "12",
          "n": 12,
          "context": "Claude Code · JSON schema"
        },
        {
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-outcomes",
          "series": "Format miss",
          "point": "Claude Sonnet 5.5 (instructions) · Claude Code",
          "metric": "structured-output-outcomes/Format miss",
          "label": "What each call produced: strict pass, format miss, wrong values or error (Format miss)",
          "value": 0,
          "unit": "count",
          "display": "0",
          "n": 12,
          "context": "Claude Code · instructions"
        },
        {
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-outcomes",
          "series": "Format miss",
          "point": "Claude Sonnet 5.5 (JSON schema) · Claude Code",
          "metric": "structured-output-outcomes/Format miss",
          "label": "What each call produced: strict pass, format miss, wrong values or error (Format miss)",
          "value": 0,
          "unit": "count",
          "display": "0",
          "n": 12,
          "context": "Claude Code · JSON schema"
        },
        {
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-outcomes",
          "series": "Wrong values",
          "point": "Claude Sonnet 5.5 (instructions) · Claude Code",
          "metric": "structured-output-outcomes/Wrong values",
          "label": "What each call produced: strict pass, format miss, wrong values or error (Wrong values)",
          "value": 0,
          "unit": "count",
          "display": "0",
          "n": 12,
          "context": "Claude Code · instructions"
        },
        {
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-outcomes",
          "series": "Wrong values",
          "point": "Claude Sonnet 5.5 (JSON schema) · Claude Code",
          "metric": "structured-output-outcomes/Wrong values",
          "label": "What each call produced: strict pass, format miss, wrong values or error (Wrong values)",
          "value": 0,
          "unit": "count",
          "display": "0",
          "n": 12,
          "context": "Claude Code · JSON schema"
        },
        {
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-outcomes",
          "series": "Error",
          "point": "Claude Sonnet 5.5 (instructions) · Claude Code",
          "metric": "structured-output-outcomes/Error",
          "label": "What each call produced: strict pass, format miss, wrong values or error (Error)",
          "value": 0,
          "unit": "count",
          "display": "0",
          "n": 12,
          "context": "Claude Code · instructions"
        },
        {
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-outcomes",
          "series": "Error",
          "point": "Claude Sonnet 5.5 (JSON schema) · Claude Code",
          "metric": "structured-output-outcomes/Error",
          "label": "What each call produced: strict pass, format miss, wrong values or error (Error)",
          "value": 0,
          "unit": "count",
          "display": "0",
          "n": 12,
          "context": "Claude Code · JSON schema"
        },
        {
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-time",
          "series": "Median time per call (the three prompts pooled)",
          "point": "Claude Sonnet 5.5 (instructions) · Claude Code",
          "metric": "structured-output-time",
          "label": "Time per call, instructions vs schema mode",
          "value": 3.52,
          "unit": "seconds",
          "display": "3.52 s",
          "n": 12,
          "range": [
            2.67,
            4.12
          ],
          "spanKind": "minmax",
          "context": "Claude Code · instructions"
        },
        {
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-time",
          "series": "Median time per call (the three prompts pooled)",
          "point": "Claude Sonnet 5.5 (JSON schema) · Claude Code",
          "metric": "structured-output-time",
          "label": "Time per call, instructions vs schema mode",
          "value": 4.2,
          "unit": "seconds",
          "display": "4.20 s",
          "n": 12,
          "range": [
            2.95,
            6.14
          ],
          "spanKind": "minmax",
          "context": "Claude Code · JSON schema"
        },
        {
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-tokens",
          "series": "Median output tokens per call",
          "point": "Claude Sonnet 5.5 (instructions) · Claude Code",
          "metric": "structured-output-tokens/Median output tokens per call",
          "label": "Output and reasoning tokens per call, instructions vs schema mode (Median output tokens per call)",
          "value": 368,
          "unit": "tokens",
          "display": "368",
          "n": 12,
          "context": "Claude Code · instructions"
        },
        {
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-tokens",
          "series": "Median output tokens per call",
          "point": "Claude Sonnet 5.5 (JSON schema) · Claude Code",
          "metric": "structured-output-tokens/Median output tokens per call",
          "label": "Output and reasoning tokens per call, instructions vs schema mode (Median output tokens per call)",
          "value": 424,
          "unit": "tokens",
          "display": "424",
          "n": 12,
          "context": "Claude Code · JSON schema"
        },
        {
          "studySlug": "prompt-cache-break-even",
          "chartId": "cache-break-even-reads",
          "series": "1-hour write (2× input), whole prefix new",
          "point": "Claude Sonnet 5.5",
          "metric": "cache-break-even-reads/1-hour write (2× input), whole prefix new",
          "label": "Reuses before a cached prefix costs less, by model and write type (calculation) (1-hour write (2× input), whole prefix new)",
          "value": 1.11,
          "unit": "score",
          "display": "1.11",
          "calculation": true,
          "context": "calculation: Agent’s recorded tokens at this model’s list price"
        },
        {
          "studySlug": "prompt-cache-break-even",
          "chartId": "cache-break-even-reads",
          "series": "1-hour write, pooled n = 6 session share, 19% already cached (as recorded)",
          "point": "Claude Sonnet 5.5",
          "metric": "cache-break-even-reads/1-hour write, pooled n = 6 session share, 19% already cached (as recorded)",
          "label": "Reuses before a cached prefix costs less, by model and write type (calculation) (1-hour write, pooled n = 6 session share, 19% already cached (as recorded))",
          "value": 0.72,
          "unit": "score",
          "display": "0.72",
          "calculation": true,
          "context": "calculation: Agent’s recorded tokens at this model’s list price"
        },
        {
          "studySlug": "prompt-cache-break-even",
          "chartId": "cache-break-even-reads",
          "series": "5-minute write (1.25× input, an assumption)",
          "point": "Claude Sonnet 5.5",
          "metric": "cache-break-even-reads/5-minute write (1.25× input, an assumption)",
          "label": "Reuses before a cached prefix costs less, by model and write type (calculation) (5-minute write (1.25× input, an assumption))",
          "value": 0.28,
          "unit": "score",
          "display": "0.28",
          "calculation": true,
          "context": "calculation: Agent’s recorded tokens at this model’s list price"
        },
        {
          "studySlug": "prompt-cache-break-even",
          "chartId": "cache-break-even-cost-curve",
          "series": "Claude Sonnet 5.5 (no cache)",
          "point": "1 turn",
          "metric": "cache-break-even-cost-curve/1 turn",
          "label": "Cost of a reused prefix with and without the cache, by session length (calculation): 1 turn",
          "value": 15.66,
          "unit": "usd",
          "display": "$15.66",
          "calculation": true,
          "context": "no cache"
        },
        {
          "studySlug": "prompt-cache-break-even",
          "chartId": "cache-break-even-cost-curve",
          "series": "Claude Sonnet 5.5 (no cache)",
          "point": "2 turns",
          "metric": "cache-break-even-cost-curve/2 turns",
          "label": "Cost of a reused prefix with and without the cache, by session length (calculation): 2 turns",
          "value": 31.32,
          "unit": "usd",
          "display": "$31.32",
          "calculation": true,
          "context": "no cache"
        },
        {
          "studySlug": "prompt-cache-break-even",
          "chartId": "cache-break-even-cost-curve",
          "series": "Claude Sonnet 5.5 (no cache)",
          "point": "3 turns",
          "metric": "cache-break-even-cost-curve/3 turns",
          "label": "Cost of a reused prefix with and without the cache, by session length (calculation): 3 turns",
          "value": 46.99,
          "unit": "usd",
          "display": "$46.99",
          "calculation": true,
          "context": "no cache"
        },
        {
          "studySlug": "prompt-cache-break-even",
          "chartId": "cache-break-even-cost-curve",
          "series": "Claude Sonnet 5.5 (no cache)",
          "point": "5 turns",
          "metric": "cache-break-even-cost-curve/5 turns",
          "label": "Cost of a reused prefix with and without the cache, by session length (calculation): 5 turns",
          "value": 78.31,
          "unit": "usd",
          "display": "$78.31",
          "calculation": true,
          "context": "no cache"
        },
        {
          "studySlug": "prompt-cache-break-even",
          "chartId": "cache-break-even-cost-curve",
          "series": "Claude Sonnet 5.5 (no cache)",
          "point": "10 turns",
          "metric": "cache-break-even-cost-curve/10 turns",
          "label": "Cost of a reused prefix with and without the cache, by session length (calculation): 10 turns",
          "value": 156.62,
          "unit": "usd",
          "display": "$156.62",
          "calculation": true,
          "context": "no cache"
        },
        {
          "studySlug": "prompt-cache-break-even",
          "chartId": "cache-break-even-cost-curve",
          "series": "Claude Sonnet 5.5 (no cache)",
          "point": "20 turns",
          "metric": "cache-break-even-cost-curve/20 turns",
          "label": "Cost of a reused prefix with and without the cache, by session length (calculation): 20 turns",
          "value": 313.24,
          "unit": "usd",
          "display": "$313.24",
          "calculation": true,
          "context": "no cache"
        },
        {
          "studySlug": "prompt-cache-break-even",
          "chartId": "cache-break-even-cost-curve",
          "series": "Claude Sonnet 5.5 (1-hour cache write)",
          "point": "1 turn",
          "metric": "cache-break-even-cost-curve/1 turn",
          "label": "Cost of a reused prefix with and without the cache, by session length (calculation): 1 turn",
          "value": 31.32,
          "unit": "usd",
          "display": "$31.32",
          "calculation": true,
          "context": "1-hour cache write"
        },
        {
          "studySlug": "prompt-cache-break-even",
          "chartId": "cache-break-even-cost-curve",
          "series": "Claude Sonnet 5.5 (1-hour cache write)",
          "point": "2 turns",
          "metric": "cache-break-even-cost-curve/2 turns",
          "label": "Cost of a reused prefix with and without the cache, by session length (calculation): 2 turns",
          "value": 32.89,
          "unit": "usd",
          "display": "$32.89",
          "calculation": true,
          "context": "1-hour cache write"
        },
        {
          "studySlug": "prompt-cache-break-even",
          "chartId": "cache-break-even-cost-curve",
          "series": "Claude Sonnet 5.5 (1-hour cache write)",
          "point": "3 turns",
          "metric": "cache-break-even-cost-curve/3 turns",
          "label": "Cost of a reused prefix with and without the cache, by session length (calculation): 3 turns",
          "value": 34.46,
          "unit": "usd",
          "display": "$34.46",
          "calculation": true,
          "context": "1-hour cache write"
        },
        {
          "studySlug": "prompt-cache-break-even",
          "chartId": "cache-break-even-cost-curve",
          "series": "Claude Sonnet 5.5 (1-hour cache write)",
          "point": "5 turns",
          "metric": "cache-break-even-cost-curve/5 turns",
          "label": "Cost of a reused prefix with and without the cache, by session length (calculation): 5 turns",
          "value": 37.59,
          "unit": "usd",
          "display": "$37.59",
          "calculation": true,
          "context": "1-hour cache write"
        },
        {
          "studySlug": "prompt-cache-break-even",
          "chartId": "cache-break-even-cost-curve",
          "series": "Claude Sonnet 5.5 (1-hour cache write)",
          "point": "10 turns",
          "metric": "cache-break-even-cost-curve/10 turns",
          "label": "Cost of a reused prefix with and without the cache, by session length (calculation): 10 turns",
          "value": 45.42,
          "unit": "usd",
          "display": "$45.42",
          "calculation": true,
          "context": "1-hour cache write"
        },
        {
          "studySlug": "prompt-cache-break-even",
          "chartId": "cache-break-even-cost-curve",
          "series": "Claude Sonnet 5.5 (1-hour cache write)",
          "point": "20 turns",
          "metric": "cache-break-even-cost-curve/20 turns",
          "label": "Cost of a reused prefix with and without the cache, by session length (calculation): 20 turns",
          "value": 61.08,
          "unit": "usd",
          "display": "$61.08",
          "calculation": true,
          "context": "1-hour cache write"
        },
        {
          "studySlug": "prompt-cache-break-even",
          "chartId": "cache-break-even-session-split",
          "series": "No cache (the same either way)",
          "point": "Claude Sonnet 5.5",
          "metric": "cache-break-even-session-split/No cache (the same either way)",
          "label": "One 10-turn session or ten 1-turn sessions: cost with and without the cache (calculation) (No cache (the same either way))",
          "value": 156.62,
          "unit": "usd",
          "display": "$156.62",
          "calculation": true,
          "context": ""
        },
        {
          "studySlug": "prompt-cache-break-even",
          "chartId": "cache-break-even-session-split",
          "series": "One 10-turn session, 1-hour cache",
          "point": "Claude Sonnet 5.5",
          "metric": "cache-break-even-session-split/One 10-turn session, 1-hour cache",
          "label": "One 10-turn session or ten 1-turn sessions: cost with and without the cache (calculation) (One 10-turn session, 1-hour cache)",
          "value": 45.42,
          "unit": "usd",
          "display": "$45.42",
          "calculation": true,
          "context": ""
        },
        {
          "studySlug": "prompt-cache-break-even",
          "chartId": "cache-break-even-session-split",
          "series": "Ten 1-turn sessions, 1-hour cache (assumed no reuse of the new prefix)",
          "point": "Claude Sonnet 5.5",
          "metric": "cache-break-even-session-split/Ten 1-turn sessions, 1-hour cache (assumed no reuse of the new prefix)",
          "label": "One 10-turn session or ten 1-turn sessions: cost with and without the cache (calculation) (Ten 1-turn sessions, 1-hour cache (assumed no reuse of the new prefix))",
          "value": 313.24,
          "unit": "usd",
          "display": "$313.24",
          "calculation": true,
          "context": ""
        },
        {
          "studySlug": "routing-holdout",
          "chartId": "routing-holdout-exact",
          "series": "Exact rate",
          "point": "Claude Sonnet 5.5 (low) · Claude Code",
          "metric": "routing-holdout-exact",
          "label": "Unseen routing decisions answered exactly right",
          "value": 0.875,
          "unit": "rate",
          "display": "88% (49/56)",
          "n": 56,
          "ci": [
            0.7637,
            0.9381
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Code · effort low"
        },
        {
          "studySlug": "routing-holdout",
          "chartId": "routing-holdout-key-accuracy",
          "series": "Key accuracy",
          "point": "Claude Sonnet 5.5 (low) · Claude Code",
          "metric": "routing-holdout-key-accuracy",
          "label": "Per-question accuracy on unseen decisions",
          "value": 0.92,
          "unit": "rate",
          "display": "92% (115/125)",
          "n": 125,
          "ci": [
            0.859,
            0.956
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Code · effort low"
        },
        {
          "studySlug": "routing-holdout",
          "chartId": "routing-holdout-by-purpose",
          "series": "Claude Sonnet 5.5 (low) · Claude Code",
          "point": "Failure class",
          "metric": "routing-holdout-by-purpose/Failure class",
          "label": "Exact rate on unseen decisions, by decision type: Failure class",
          "value": 1,
          "unit": "rate",
          "display": "100% (14/14)",
          "n": 14,
          "ci": [
            0.7847,
            1
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Code · effort low"
        },
        {
          "studySlug": "routing-holdout",
          "chartId": "routing-holdout-by-purpose",
          "series": "Claude Sonnet 5.5 (low) · Claude Code",
          "point": "Message intent",
          "metric": "routing-holdout-by-purpose/Message intent",
          "label": "Exact rate on unseen decisions, by decision type: Message intent",
          "value": 1,
          "unit": "rate",
          "display": "100% (14/14)",
          "n": 14,
          "ci": [
            0.7847,
            1
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Code · effort low"
        },
        {
          "studySlug": "routing-holdout",
          "chartId": "routing-holdout-by-purpose",
          "series": "Claude Sonnet 5.5 (low) · Claude Code",
          "point": "Is it a rule?",
          "metric": "routing-holdout-by-purpose/Is it a rule?",
          "label": "Exact rate on unseen decisions, by decision type: Is it a rule?",
          "value": 0.9286,
          "unit": "rate",
          "display": "93% (13/14)",
          "n": 14,
          "ci": [
            0.6853,
            0.9873
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Code · effort low"
        },
        {
          "studySlug": "routing-holdout",
          "chartId": "routing-holdout-by-purpose",
          "series": "Claude Sonnet 5.5 (low) · Claude Code",
          "point": "Context shape",
          "metric": "routing-holdout-by-purpose/Context shape",
          "label": "Exact rate on unseen decisions, by decision type: Context shape",
          "value": 0.5714,
          "unit": "rate",
          "display": "57% (8/14)",
          "n": 14,
          "ci": [
            0.3259,
            0.7862
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Code · effort low"
        },
        {
          "studySlug": "routing-holdout",
          "chartId": "routing-holdout-tuned-vs-unseen",
          "series": "Tuned set (routing-jev-vs-llm)",
          "point": "Claude Sonnet 5.5 (low) · Claude Code",
          "metric": "routing-holdout-tuned-vs-unseen/Tuned set (routing-jev-vs-llm)",
          "label": "Tuned case set vs unseen holdout: exact rate per router (Tuned set (routing-jev-vs-llm))",
          "value": 0.939,
          "unit": "rate",
          "display": "94% (77/82)",
          "n": 82,
          "ci": [
            0.8651,
            0.9737
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Code · effort low"
        },
        {
          "studySlug": "routing-holdout",
          "chartId": "routing-holdout-tuned-vs-unseen",
          "series": "Unseen holdout",
          "point": "Claude Sonnet 5.5 (low) · Claude Code",
          "metric": "routing-holdout-tuned-vs-unseen/Unseen holdout",
          "label": "Tuned case set vs unseen holdout: exact rate per router (Unseen holdout)",
          "value": 0.875,
          "unit": "rate",
          "display": "88% (49/56)",
          "n": 56,
          "ci": [
            0.7637,
            0.9381
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Code · effort low"
        },
        {
          "studySlug": "routing-holdout",
          "chartId": "routing-holdout-latency",
          "series": "Wall time",
          "point": "Claude Sonnet 5.5 (low) · Claude Code",
          "metric": "routing-holdout-latency/Wall time",
          "label": "Time per routing decision, by route (Wall time)",
          "value": 2.359,
          "unit": "seconds",
          "display": "2.36 s",
          "n": 56,
          "range": [
            2.359,
            3.657
          ],
          "spanKind": "p50-p95",
          "context": "Claude Code · effort low"
        },
        {
          "studySlug": "routing-holdout",
          "chartId": "routing-holdout-latency",
          "series": "Model time (API, CLI-reported)",
          "point": "Claude Sonnet 5.5 (low) · Claude Code",
          "metric": "routing-holdout-latency/Model time (API, CLI-reported)",
          "label": "Time per routing decision, by route (Model time (API, CLI-reported))",
          "value": 1.485,
          "unit": "seconds",
          "display": "1.49 s",
          "n": 56,
          "range": [
            1.485,
            2.377
          ],
          "spanKind": "p50-p95",
          "context": "Claude Code · effort low"
        },
        {
          "studySlug": "routing-holdout",
          "chartId": "routing-holdout-cost-per-1000",
          "series": "Cost",
          "point": "Claude Sonnet 5.5 (low) · Claude Code",
          "metric": "routing-holdout-cost-per-1000",
          "label": "Cost per 1,000 unseen routing decisions",
          "value": 7.244,
          "unit": "usd",
          "display": "$7.24",
          "n": 56,
          "calculation": true,
          "context": "Claude Code · effort low"
        },
        {
          "studySlug": "routing-holdout",
          "statId": "holdout-gap-claude-sonnet",
          "metric": "stat:holdout-gap-claude-sonnet",
          "label": "Claude Sonnet 5.5 (low) · Claude Code: holdout minus tuned-set exact rate",
          "value": -0.064,
          "unit": "rate",
          "display": "−6.4 points",
          "n": 56,
          "calculation": true,
          "context": "low"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-share",
          "series": "Median call",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "thinking-bill-share",
          "label": "Reasoning share of output tokens per call on hard tasks (calculation)",
          "value": 54.54,
          "unit": "percent",
          "display": "54.5%",
          "n": 24,
          "range": [
            0,
            95.91
          ],
          "spanKind": "minmax",
          "calculation": true,
          "polarity": "none",
          "context": "Claude Code"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-cost-per-call",
          "series": "Reasoning (output tokens)",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "thinking-bill-cost-per-call/Reasoning (output tokens)",
          "label": "List-price cost per call: reasoning, remaining output and input (calculation) (Reasoning (output tokens))",
          "value": 0.006665,
          "unit": "usd",
          "display": "$0.0067",
          "n": 24,
          "calculation": true,
          "polarity": "none",
          "context": "Claude Code"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-cost-per-call",
          "series": "Remaining output (visible-answer estimate)",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "thinking-bill-cost-per-call/Remaining output (visible-answer estimate)",
          "label": "List-price cost per call: reasoning, remaining output and input (calculation) (Remaining output (visible-answer estimate))",
          "value": 0.003672,
          "unit": "usd",
          "display": "$0.0037",
          "n": 24,
          "calculation": true,
          "polarity": "none",
          "context": "Claude Code"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-cost-per-call",
          "series": "Input (prompt, cache priced)",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "thinking-bill-cost-per-call/Input (prompt, cache priced)",
          "label": "List-price cost per call: reasoning, remaining output and input (calculation) (Input (prompt, cache priced))",
          "value": 0.004012,
          "unit": "usd",
          "display": "$0.0040",
          "n": 24,
          "calculation": true,
          "polarity": "none",
          "context": "Claude Code"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-by-effort",
          "series": "Reasoning cost per strict pass",
          "point": "Claude Sonnet 5.5 (low) · Claude Code",
          "metric": "thinking-bill-by-effort/Reasoning cost per strict pass",
          "label": "Reasoning cost per strict pass by effort, with the total (calculation) (Reasoning cost per strict pass)",
          "value": 0.004317,
          "unit": "usd",
          "display": "$0.0043",
          "n": 16,
          "calculation": true,
          "polarity": "none",
          "context": "Claude Code · effort low"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-by-effort",
          "series": "Reasoning cost per strict pass",
          "point": "Claude Sonnet 5.5 (medium) · Claude Code",
          "metric": "thinking-bill-by-effort/Reasoning cost per strict pass",
          "label": "Reasoning cost per strict pass by effort, with the total (calculation) (Reasoning cost per strict pass)",
          "value": 0.005946,
          "unit": "usd",
          "display": "$0.0059",
          "n": 16,
          "calculation": true,
          "polarity": "none",
          "context": "Claude Code · effort medium"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-by-effort",
          "series": "Reasoning cost per strict pass",
          "point": "Claude Sonnet 5.5 (high) · Claude Code",
          "metric": "thinking-bill-by-effort/Reasoning cost per strict pass",
          "label": "Reasoning cost per strict pass by effort, with the total (calculation) (Reasoning cost per strict pass)",
          "value": 0.009369,
          "unit": "usd",
          "display": "$0.0094",
          "n": 16,
          "calculation": true,
          "polarity": "none",
          "context": "Claude Code · effort high"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-by-effort",
          "series": "Reasoning cost per strict pass",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "thinking-bill-by-effort/Reasoning cost per strict pass",
          "label": "Reasoning cost per strict pass by effort, with the total (calculation) (Reasoning cost per strict pass)",
          "value": 0.006299,
          "unit": "usd",
          "display": "$0.0063",
          "n": 16,
          "calculation": true,
          "polarity": "none",
          "context": "Claude Code"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-by-effort",
          "series": "Total cost per strict pass",
          "point": "Claude Sonnet 5.5 (low) · Claude Code",
          "metric": "thinking-bill-by-effort/Total cost per strict pass",
          "label": "Reasoning cost per strict pass by effort, with the total (calculation) (Total cost per strict pass)",
          "value": 0.012191,
          "unit": "usd",
          "display": "$0.012",
          "n": 16,
          "calculation": true,
          "polarity": "none",
          "context": "Claude Code · effort low"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-by-effort",
          "series": "Total cost per strict pass",
          "point": "Claude Sonnet 5.5 (medium) · Claude Code",
          "metric": "thinking-bill-by-effort/Total cost per strict pass",
          "label": "Reasoning cost per strict pass by effort, with the total (calculation) (Total cost per strict pass)",
          "value": 0.01352,
          "unit": "usd",
          "display": "$0.014",
          "n": 16,
          "calculation": true,
          "polarity": "none",
          "context": "Claude Code · effort medium"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-by-effort",
          "series": "Total cost per strict pass",
          "point": "Claude Sonnet 5.5 (high) · Claude Code",
          "metric": "thinking-bill-by-effort/Total cost per strict pass",
          "label": "Reasoning cost per strict pass by effort, with the total (calculation) (Total cost per strict pass)",
          "value": 0.016705,
          "unit": "usd",
          "display": "$0.017",
          "n": 16,
          "calculation": true,
          "polarity": "none",
          "context": "Claude Code · effort high"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-by-effort",
          "series": "Total cost per strict pass",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "thinking-bill-by-effort/Total cost per strict pass",
          "label": "Reasoning cost per strict pass by effort, with the total (calculation) (Total cost per strict pass)",
          "value": 0.013978,
          "unit": "usd",
          "display": "$0.014",
          "n": 16,
          "calculation": true,
          "polarity": "none",
          "context": "Claude Code"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-short-vs-hard",
          "series": "Eight hard tasks",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "thinking-bill-short-vs-hard/Eight hard tasks",
          "label": "Reasoning share on short tasks vs hard tasks (calculation) (Eight hard tasks)",
          "value": 54.54,
          "unit": "percent",
          "display": "54.5%",
          "n": 24,
          "range": [
            0,
            95.91
          ],
          "spanKind": "minmax",
          "calculation": true,
          "polarity": "none",
          "context": "Claude Code"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-short-vs-hard",
          "series": "Five short tasks",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "thinking-bill-short-vs-hard/Five short tasks",
          "label": "Reasoning share on short tasks vs hard tasks (calculation) (Five short tasks)",
          "value": 0,
          "unit": "percent",
          "display": "0%",
          "n": 15,
          "range": [
            0,
            72.75
          ],
          "spanKind": "minmax",
          "calculation": true,
          "polarity": "none",
          "context": "Claude Code"
        },
        {
          "studySlug": "llm-speed-anatomy",
          "chartId": "speed-anatomy-first-text",
          "series": "Time to first text",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "speed-anatomy-first-text",
          "label": "Time to first text: a 250-line answer, six models",
          "value": 1.96,
          "unit": "seconds",
          "display": "1.96 s",
          "n": 4,
          "range": [
            0.88,
            4.09
          ],
          "spanKind": "minmax",
          "context": "Claude Code"
        },
        {
          "studySlug": "llm-speed-anatomy",
          "chartId": "speed-anatomy-output-speed",
          "series": "Visible tokens per second",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "speed-anatomy-output-speed",
          "label": "Output speed after the first text: visible tokens per second (calculation)",
          "value": 231.7,
          "unit": "tokens",
          "display": "232",
          "n": 4,
          "range": [
            230.3,
            233
          ],
          "spanKind": "minmax",
          "calculation": true,
          "context": "Claude Code"
        },
        {
          "studySlug": "llm-speed-anatomy",
          "chartId": "speed-anatomy-chars-per-second",
          "series": "Characters per second",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "speed-anatomy-chars-per-second",
          "label": "Output speed in characters per second after the first text (calculation)",
          "value": 517,
          "unit": "count",
          "display": "517",
          "n": 4,
          "range": [
            513,
            519
          ],
          "spanKind": "minmax",
          "calculation": true,
          "context": "Claude Code"
        },
        {
          "studySlug": "llm-speed-anatomy",
          "chartId": "speed-anatomy-prompt-size",
          "series": "Claude Sonnet 5.5 · Claude Code",
          "point": "1k",
          "metric": "speed-anatomy-prompt-size/1k",
          "label": "Time to first text as the prompt grows: 1k",
          "value": 1.45,
          "unit": "seconds",
          "display": "1.45 s",
          "n": 3,
          "range": [
            1.23,
            1.72
          ],
          "spanKind": "minmax",
          "calculation": true,
          "context": "Claude Code"
        },
        {
          "studySlug": "llm-speed-anatomy",
          "chartId": "speed-anatomy-prompt-size",
          "series": "Claude Sonnet 5.5 · Claude Code",
          "point": "16k",
          "metric": "speed-anatomy-prompt-size/16k",
          "label": "Time to first text as the prompt grows: 16k",
          "value": 1.78,
          "unit": "seconds",
          "display": "1.78 s",
          "n": 3,
          "range": [
            1.64,
            2.11
          ],
          "spanKind": "minmax",
          "calculation": true,
          "context": "Claude Code"
        },
        {
          "studySlug": "llm-speed-anatomy",
          "chartId": "speed-anatomy-prompt-size",
          "series": "Claude Sonnet 5.5 · Claude Code",
          "point": "64k",
          "metric": "speed-anatomy-prompt-size/64k",
          "label": "Time to first text as the prompt grows: 64k",
          "value": 3.07,
          "unit": "seconds",
          "display": "3.07 s",
          "n": 3,
          "range": [
            1.38,
            3.61
          ],
          "spanKind": "minmax",
          "calculation": true,
          "context": "Claude Code"
        },
        {
          "studySlug": "llm-speed-anatomy",
          "chartId": "speed-anatomy-total-by-size",
          "series": "1k prompt",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "speed-anatomy-total-by-size/1k prompt",
          "label": "Total time per call by prompt size (1k prompt)",
          "value": 1.78,
          "unit": "seconds",
          "display": "1.78 s",
          "n": 3,
          "range": [
            1.57,
            2.12
          ],
          "spanKind": "minmax",
          "context": "Claude Code"
        },
        {
          "studySlug": "llm-speed-anatomy",
          "chartId": "speed-anatomy-total-by-size",
          "series": "16k prompt",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "speed-anatomy-total-by-size/16k prompt",
          "label": "Total time per call by prompt size (16k prompt)",
          "value": 2.1,
          "unit": "seconds",
          "display": "2.10 s",
          "n": 3,
          "range": [
            1.98,
            2.48
          ],
          "spanKind": "minmax",
          "context": "Claude Code"
        },
        {
          "studySlug": "llm-speed-anatomy",
          "chartId": "speed-anatomy-total-by-size",
          "series": "64k prompt",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "speed-anatomy-total-by-size/64k prompt",
          "label": "Total time per call by prompt size (64k prompt)",
          "value": 3.44,
          "unit": "seconds",
          "display": "3.44 s",
          "n": 3,
          "range": [
            1.74,
            4.38
          ],
          "spanKind": "minmax",
          "context": "Claude Code"
        },
        {
          "studySlug": "llm-speed-anatomy",
          "chartId": "speed-anatomy-lookup-correct",
          "series": "Exact answer",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "speed-anatomy-lookup-correct",
          "label": "Exact lookup answers at the 1k, 16k and 64k prompt-size targets",
          "value": 1,
          "unit": "rate",
          "display": "100% (9/9)",
          "n": 9,
          "ci": [
            0.7009,
            1
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Code"
        },
        {
          "studySlug": "haiku-retry-or-escalate",
          "chartId": "retry-escalate-call-cost-by-task",
          "series": "Claude Sonnet 5.5 · Claude Code",
          "point": "Interval merge fix",
          "metric": "retry-escalate-call-cost-by-task/Interval merge fix",
          "label": "List-price cost of one call, Haiku 4.5 vs Sonnet 5.5, on each hard task (calculation): Interval merge fix",
          "value": 0.00557,
          "unit": "usd",
          "display": "$0.0056",
          "n": 3,
          "calculation": true,
          "context": "Claude Code"
        },
        {
          "studySlug": "haiku-retry-or-escalate",
          "chartId": "retry-escalate-call-cost-by-task",
          "series": "Claude Sonnet 5.5 · Claude Code",
          "point": "DST day length",
          "metric": "retry-escalate-call-cost-by-task/DST day length",
          "label": "List-price cost of one call, Haiku 4.5 vs Sonnet 5.5, on each hard task (calculation): DST day length",
          "value": 0.02532,
          "unit": "usd",
          "display": "$0.025",
          "n": 3,
          "calculation": true,
          "context": "Claude Code"
        },
        {
          "studySlug": "haiku-retry-or-escalate",
          "chartId": "retry-escalate-call-cost-by-task",
          "series": "Claude Sonnet 5.5 · Claude Code",
          "point": "CSV parser",
          "metric": "retry-escalate-call-cost-by-task/CSV parser",
          "label": "List-price cost of one call, Haiku 4.5 vs Sonnet 5.5, on each hard task (calculation): CSV parser",
          "value": 0.01464,
          "unit": "usd",
          "display": "$0.015",
          "n": 3,
          "calculation": true,
          "context": "Claude Code"
        },
        {
          "studySlug": "haiku-retry-or-escalate",
          "chartId": "retry-escalate-call-cost-by-task",
          "series": "Claude Sonnet 5.5 · Claude Code",
          "point": "Event-loop order",
          "metric": "retry-escalate-call-cost-by-task/Event-loop order",
          "label": "List-price cost of one call, Haiku 4.5 vs Sonnet 5.5, on each hard task (calculation): Event-loop order",
          "value": 0.01588,
          "unit": "usd",
          "display": "$0.016",
          "n": 3,
          "calculation": true,
          "context": "Claude Code"
        },
        {
          "studySlug": "haiku-retry-or-escalate",
          "chartId": "retry-escalate-call-cost-by-task",
          "series": "Claude Sonnet 5.5 · Claude Code",
          "point": "Room schedule",
          "metric": "retry-escalate-call-cost-by-task/Room schedule",
          "label": "List-price cost of one call, Haiku 4.5 vs Sonnet 5.5, on each hard task (calculation): Room schedule",
          "value": 0.01243,
          "unit": "usd",
          "display": "$0.012",
          "n": 3,
          "calculation": true,
          "context": "Claude Code"
        },
        {
          "studySlug": "haiku-retry-or-escalate",
          "chartId": "retry-escalate-call-cost-by-task",
          "series": "Claude Sonnet 5.5 · Claude Code",
          "point": "SemVer regex",
          "metric": "retry-escalate-call-cost-by-task/SemVer regex",
          "label": "List-price cost of one call, Haiku 4.5 vs Sonnet 5.5, on each hard task (calculation): SemVer regex",
          "value": 0.00514,
          "unit": "usd",
          "display": "$0.0051",
          "n": 3,
          "calculation": true,
          "context": "Claude Code"
        },
        {
          "studySlug": "haiku-retry-or-escalate",
          "chartId": "retry-escalate-call-cost-by-task",
          "series": "Claude Sonnet 5.5 · Claude Code",
          "point": "Money refactor",
          "metric": "retry-escalate-call-cost-by-task/Money refactor",
          "label": "List-price cost of one call, Haiku 4.5 vs Sonnet 5.5, on each hard task (calculation): Money refactor",
          "value": 0.00961,
          "unit": "usd",
          "display": "$0.0096",
          "n": 3,
          "calculation": true,
          "context": "Claude Code"
        },
        {
          "studySlug": "haiku-retry-or-escalate",
          "chartId": "retry-escalate-call-cost-by-task",
          "series": "Claude Sonnet 5.5 · Claude Code",
          "point": "SQLite report query",
          "metric": "retry-escalate-call-cost-by-task/SQLite report query",
          "label": "List-price cost of one call, Haiku 4.5 vs Sonnet 5.5, on each hard task (calculation): SQLite report query",
          "value": 0.01719,
          "unit": "usd",
          "display": "$0.017",
          "n": 3,
          "calculation": true,
          "context": "Claude Code"
        },
        {
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-pass-rate",
          "series": "Strict pass",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "harder-h2h-pass-rate/Strict pass",
          "label": "Pass rate on 4 harder tasks (Strict pass)",
          "value": 0.375,
          "unit": "rate",
          "display": "38% (6/16)",
          "n": 16,
          "ci": [
            0.1848,
            0.6136
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Code"
        },
        {
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-pass-rate",
          "series": "Lenient (format misses counted)",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "harder-h2h-pass-rate/Lenient (format misses counted)",
          "label": "Pass rate on 4 harder tasks (Lenient (format misses counted))",
          "value": 0.375,
          "unit": "rate",
          "display": "38% (6/16)",
          "n": 16,
          "ci": [
            0.1848,
            0.6136
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Code"
        },
        {
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-tool-attempts",
          "series": "Tool attempt",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "harder-h2h-tool-attempts",
          "label": "Calls that tried a tool although tools were off",
          "value": 0.3125,
          "unit": "rate",
          "display": "31% (5/16)",
          "n": 16,
          "ci": [
            0.1416,
            0.556
          ],
          "spanKind": "ci95",
          "polarity": "none",
          "context": "Claude Code"
        },
        {
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-pass-by-task",
          "series": "Claude Sonnet 5.5 · Claude Code",
          "point": "10x10 nonogram",
          "metric": "harder-h2h-pass-by-task/10x10 nonogram",
          "label": "Strict pass rate by task: 10x10 nonogram",
          "value": 1,
          "unit": "rate",
          "display": "100% (4/4)",
          "n": 4,
          "ci": [
            0.5101,
            1
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Code"
        },
        {
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-pass-by-task",
          "series": "Claude Sonnet 5.5 · Claude Code",
          "point": "Sudoku, 22 givens",
          "metric": "harder-h2h-pass-by-task/Sudoku, 22 givens",
          "label": "Strict pass rate by task: Sudoku, 22 givens",
          "value": 0,
          "unit": "rate",
          "display": "0% (0/4)",
          "n": 4,
          "ci": [
            0,
            0.4899
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Code"
        },
        {
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-pass-by-task",
          "series": "Claude Sonnet 5.5 · Claude Code",
          "point": "6x6 Skyscrapers",
          "metric": "harder-h2h-pass-by-task/6x6 Skyscrapers",
          "label": "Strict pass rate by task: 6x6 Skyscrapers",
          "value": 0,
          "unit": "rate",
          "display": "0% (0/4)",
          "n": 4,
          "ci": [
            0,
            0.4899
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Code"
        },
        {
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-pass-by-task",
          "series": "Claude Sonnet 5.5 · Claude Code",
          "point": "Seeded shuffle output",
          "metric": "harder-h2h-pass-by-task/Seeded shuffle output",
          "label": "Strict pass rate by task: Seeded shuffle output",
          "value": 0.5,
          "unit": "rate",
          "display": "50% (2/4)",
          "n": 4,
          "ci": [
            0.15,
            0.85
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Code"
        },
        {
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-total-latency",
          "series": "Total time per call",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "harder-h2h-total-latency",
          "label": "Total time per call on harder tasks",
          "value": 70.43,
          "unit": "seconds",
          "display": "70.4 s",
          "n": 12,
          "range": [
            4.32,
            210.08
          ],
          "spanKind": "minmax",
          "context": "Claude Code"
        },
        {
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-output-tokens",
          "series": "Output tokens",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "harder-h2h-output-tokens/Output tokens",
          "label": "Output tokens per call on harder tasks (Output tokens)",
          "value": 9287,
          "unit": "tokens",
          "display": "9,287",
          "n": 12,
          "range": [
            407,
            27921
          ],
          "spanKind": "minmax",
          "polarity": "none",
          "context": "Claude Code"
        },
        {
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-cost-per-pass",
          "series": "Cost per strict pass",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "harder-h2h-cost-per-pass",
          "label": "List-price cost per strict pass on harder tasks (calculation)",
          "value": 0.23843,
          "unit": "usd",
          "display": "$0.24",
          "n": 16,
          "calculation": true,
          "context": "Claude Code"
        }
      ]
    },
    {
      "slug": "claude-opus-5-5",
      "name": "Claude Opus 5.5",
      "vendor": "Anthropic",
      "kind": "model",
      "description": "Anthropic’s large Claude model, measured through Claude Code at low, medium, high and default effort.",
      "aliases": [
        "Claude Opus 5.5",
        "Opus 5.5"
      ],
      "facts": [
        {
          "studySlug": "swe-bench-opus-vs-sonnet",
          "chartId": "swebench-opus-sonnet-resolved",
          "series": "Resolved",
          "point": "Claude Opus 5.5 (Agent, new build)",
          "metric": "swebench-opus-sonnet-resolved",
          "label": "Resolved on the same 3 SWE-bench Verified instances (interim)",
          "value": 0.6667,
          "unit": "rate",
          "display": "67% (2/3)",
          "n": 3,
          "ci": [
            0.2077,
            0.9385
          ],
          "spanKind": "ci95",
          "context": "Agent · new build · SWE-bench Verified, interim paired probe"
        },
        {
          "studySlug": "swe-bench-opus-vs-sonnet",
          "chartId": "swebench-opus-sonnet-cost-per-attempt",
          "series": "List-price cost per attempt",
          "point": "Claude Opus 5.5 (Agent, new build)",
          "metric": "swebench-opus-sonnet-cost-per-attempt",
          "label": "List-price cost per attempt (calculation)",
          "value": 7.59,
          "unit": "usd",
          "display": "$7.59",
          "n": 3,
          "calculation": true,
          "context": "Agent · new build · SWE-bench Verified, interim paired probe"
        },
        {
          "studySlug": "swe-bench-opus-vs-sonnet",
          "chartId": "swebench-opus-sonnet-minutes",
          "series": "Worker minutes per attempt",
          "point": "Claude Opus 5.5 (Agent, new build)",
          "metric": "swebench-opus-sonnet-minutes",
          "label": "Worker time per attempt",
          "value": 20.26,
          "unit": "minutes",
          "display": "20.3 min",
          "n": 3,
          "range": [
            9.76,
            25.29
          ],
          "spanKind": "minmax",
          "context": "Agent · new build · SWE-bench Verified, interim paired probe"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-pass-rate",
          "series": "Pass rate",
          "point": "Claude Opus 5.5 (high) · Claude Code",
          "metric": "h2h-pass-rate",
          "label": "Pass rate on five validated tasks",
          "value": 1,
          "unit": "rate",
          "display": "100% (15/15)",
          "n": 15,
          "ci": [
            0.7961,
            1
          ],
          "spanKind": "ci95",
          "context": "Claude Code · effort high · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-pass-rate",
          "series": "Pass rate",
          "point": "Claude Opus 5.5 · Claude Code",
          "metric": "h2h-pass-rate",
          "label": "Pass rate on five validated tasks",
          "value": 1,
          "unit": "rate",
          "display": "100% (15/15)",
          "n": 15,
          "ci": [
            0.7961,
            1
          ],
          "spanKind": "ci95",
          "context": "Claude Code · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-pass-rate",
          "series": "Pass rate",
          "point": "Claude Opus 5.5 (low) · Claude Code",
          "metric": "h2h-pass-rate",
          "label": "Pass rate on five validated tasks",
          "value": 1,
          "unit": "rate",
          "display": "100% (15/15)",
          "n": 15,
          "ci": [
            0.7961,
            1
          ],
          "spanKind": "ci95",
          "context": "Claude Code · effort low · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-total-latency",
          "series": "Total time per call",
          "point": "Claude Opus 5.5 (high) · Claude Code",
          "metric": "h2h-total-latency",
          "label": "Total time per call",
          "value": 2.71,
          "unit": "seconds",
          "display": "2.71 s",
          "n": 15,
          "range": [
            2.45,
            11.78
          ],
          "spanKind": "minmax",
          "context": "Claude Code · effort high · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-total-latency",
          "series": "Total time per call",
          "point": "Claude Opus 5.5 · Claude Code",
          "metric": "h2h-total-latency",
          "label": "Total time per call",
          "value": 2.75,
          "unit": "seconds",
          "display": "2.75 s",
          "n": 15,
          "range": [
            2.47,
            8.91
          ],
          "spanKind": "minmax",
          "context": "Claude Code · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-total-latency",
          "series": "Total time per call",
          "point": "Claude Opus 5.5 (low) · Claude Code",
          "metric": "h2h-total-latency",
          "label": "Total time per call",
          "value": 2.83,
          "unit": "seconds",
          "display": "2.83 s",
          "n": 15,
          "range": [
            2.35,
            6.62
          ],
          "spanKind": "minmax",
          "context": "Claude Code · effort low · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-first-useful-latency",
          "series": "Time to first useful output",
          "point": "Claude Opus 5.5 (high) · Claude Code",
          "metric": "h2h-first-useful-latency",
          "label": "Time to first useful output",
          "value": 2.04,
          "unit": "seconds",
          "display": "2.04 s",
          "n": 15,
          "range": [
            1.4,
            9.94
          ],
          "spanKind": "minmax",
          "context": "Claude Code · effort high · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-first-useful-latency",
          "series": "Time to first useful output",
          "point": "Claude Opus 5.5 · Claude Code",
          "metric": "h2h-first-useful-latency",
          "label": "Time to first useful output",
          "value": 1.92,
          "unit": "seconds",
          "display": "1.92 s",
          "n": 15,
          "range": [
            1.56,
            7.23
          ],
          "spanKind": "minmax",
          "context": "Claude Code · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-first-useful-latency",
          "series": "Time to first useful output",
          "point": "Claude Opus 5.5 (low) · Claude Code",
          "metric": "h2h-first-useful-latency",
          "label": "Time to first useful output",
          "value": 2.39,
          "unit": "seconds",
          "display": "2.39 s",
          "n": 15,
          "range": [
            1.45,
            4.9
          ],
          "spanKind": "minmax",
          "context": "Claude Code · effort low · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-input-tokens",
          "series": "Cache read",
          "point": "Claude Opus 5.5 (high) · Claude Code",
          "metric": "h2h-input-tokens/Cache read",
          "label": "Input tokens per call: what the CLI sends (Cache read)",
          "value": 1463,
          "unit": "tokens",
          "display": "1,463",
          "n": 15,
          "context": "Claude Code · effort high · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-input-tokens",
          "series": "Cache read",
          "point": "Claude Opus 5.5 · Claude Code",
          "metric": "h2h-input-tokens/Cache read",
          "label": "Input tokens per call: what the CLI sends (Cache read)",
          "value": 1401,
          "unit": "tokens",
          "display": "1,401",
          "n": 15,
          "context": "Claude Code · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-input-tokens",
          "series": "Cache read",
          "point": "Claude Opus 5.5 (low) · Claude Code",
          "metric": "h2h-input-tokens/Cache read",
          "label": "Input tokens per call: what the CLI sends (Cache read)",
          "value": 1463,
          "unit": "tokens",
          "display": "1,463",
          "n": 15,
          "context": "Claude Code · effort low · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-input-tokens",
          "series": "Other input",
          "point": "Claude Opus 5.5 (high) · Claude Code",
          "metric": "h2h-input-tokens/Other input",
          "label": "Input tokens per call: what the CLI sends (Other input)",
          "value": 619,
          "unit": "tokens",
          "display": "619",
          "n": 15,
          "context": "Claude Code · effort high · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-input-tokens",
          "series": "Other input",
          "point": "Claude Opus 5.5 · Claude Code",
          "metric": "h2h-input-tokens/Other input",
          "label": "Input tokens per call: what the CLI sends (Other input)",
          "value": 680,
          "unit": "tokens",
          "display": "680",
          "n": 15,
          "context": "Claude Code · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-input-tokens",
          "series": "Other input",
          "point": "Claude Opus 5.5 (low) · Claude Code",
          "metric": "h2h-input-tokens/Other input",
          "label": "Input tokens per call: what the CLI sends (Other input)",
          "value": 618,
          "unit": "tokens",
          "display": "618",
          "n": 15,
          "context": "Claude Code · effort low · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-output-tokens",
          "series": "Output tokens",
          "point": "Claude Opus 5.5 (high) · Claude Code",
          "metric": "h2h-output-tokens/Output tokens",
          "label": "Output tokens per call (Output tokens)",
          "value": 78,
          "unit": "tokens",
          "display": "78",
          "n": 15,
          "context": "Claude Code · effort high · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-output-tokens",
          "series": "Output tokens",
          "point": "Claude Opus 5.5 · Claude Code",
          "metric": "h2h-output-tokens/Output tokens",
          "label": "Output tokens per call (Output tokens)",
          "value": 64,
          "unit": "tokens",
          "display": "64",
          "n": 15,
          "context": "Claude Code · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-output-tokens",
          "series": "Output tokens",
          "point": "Claude Opus 5.5 (low) · Claude Code",
          "metric": "h2h-output-tokens/Output tokens",
          "label": "Output tokens per call (Output tokens)",
          "value": 64,
          "unit": "tokens",
          "display": "64",
          "n": 15,
          "context": "Claude Code · effort low · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-list-price-per-call",
          "series": "Cost per call",
          "point": "Claude Opus 5.5 (high) · Claude Code",
          "metric": "h2h-list-price-per-call",
          "label": "List-price cost per call (calculation)",
          "value": 0.00694,
          "unit": "usd",
          "display": "$0.0069",
          "n": 15,
          "range": [
            0.00592,
            0.02708
          ],
          "spanKind": "minmax",
          "calculation": true,
          "context": "Claude Code · effort high · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-list-price-per-call",
          "series": "Cost per call",
          "point": "Claude Opus 5.5 · Claude Code",
          "metric": "h2h-list-price-per-call",
          "label": "List-price cost per call (calculation)",
          "value": 0.00688,
          "unit": "usd",
          "display": "$0.0069",
          "n": 15,
          "range": [
            0.00592,
            0.02226
          ],
          "spanKind": "minmax",
          "calculation": true,
          "context": "Claude Code · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-list-price-per-call",
          "series": "Cost per call",
          "point": "Claude Opus 5.5 (low) · Claude Code",
          "metric": "h2h-list-price-per-call",
          "label": "List-price cost per call (calculation)",
          "value": 0.00688,
          "unit": "usd",
          "display": "$0.0069",
          "n": 15,
          "range": [
            0.00582,
            0.01793
          ],
          "spanKind": "minmax",
          "calculation": true,
          "context": "Claude Code · effort low · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-cost-per-pass",
          "series": "Cost per pass",
          "point": "Claude Opus 5.5 (low) · Claude Code",
          "metric": "h2h-cost-per-pass",
          "label": "List-price cost per passing answer (calculation)",
          "value": 0.00829,
          "unit": "usd",
          "display": "$0.0083",
          "n": 15,
          "calculation": true,
          "context": "Claude Code · effort low · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-cost-per-pass",
          "series": "Cost per pass",
          "point": "Claude Opus 5.5 · Claude Code",
          "metric": "h2h-cost-per-pass",
          "label": "List-price cost per passing answer (calculation)",
          "value": 0.01009,
          "unit": "usd",
          "display": "$0.010",
          "n": 15,
          "calculation": true,
          "context": "Claude Code · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-cost-per-pass",
          "series": "Cost per pass",
          "point": "Claude Opus 5.5 (high) · Claude Code",
          "metric": "h2h-cost-per-pass",
          "label": "List-price cost per passing answer (calculation)",
          "value": 0.01049,
          "unit": "usd",
          "display": "$0.010",
          "n": 15,
          "calculation": true,
          "context": "Claude Code · effort high · five short validated tasks"
        },
        {
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-pass-rate",
          "series": "Strict pass",
          "point": "Claude Opus 5.5 · Claude Code",
          "metric": "hard-h2h-pass-rate/Strict pass",
          "label": "Pass rate on eight hard tasks (Strict pass)",
          "value": 1,
          "unit": "rate",
          "display": "100% (24/24)",
          "n": 24,
          "ci": [
            0.862,
            1
          ],
          "spanKind": "ci95",
          "context": "Claude Code · eight hard validated tasks"
        },
        {
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-pass-rate",
          "series": "Strict pass",
          "point": "Claude Opus 5.5 (high) · Claude Code",
          "metric": "hard-h2h-pass-rate/Strict pass",
          "label": "Pass rate on eight hard tasks (Strict pass)",
          "value": 1,
          "unit": "rate",
          "display": "100% (24/24)",
          "n": 24,
          "ci": [
            0.862,
            1
          ],
          "spanKind": "ci95",
          "context": "Claude Code · effort high · eight hard validated tasks"
        },
        {
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-pass-rate",
          "series": "Lenient (format misses counted)",
          "point": "Claude Opus 5.5 · Claude Code",
          "metric": "hard-h2h-pass-rate/Lenient (format misses counted)",
          "label": "Pass rate on eight hard tasks (Lenient (format misses counted))",
          "value": 1,
          "unit": "rate",
          "display": "100% (24/24)",
          "n": 24,
          "ci": [
            0.862,
            1
          ],
          "spanKind": "ci95",
          "context": "Claude Code · eight hard validated tasks"
        },
        {
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-pass-rate",
          "series": "Lenient (format misses counted)",
          "point": "Claude Opus 5.5 (high) · Claude Code",
          "metric": "hard-h2h-pass-rate/Lenient (format misses counted)",
          "label": "Pass rate on eight hard tasks (Lenient (format misses counted))",
          "value": 1,
          "unit": "rate",
          "display": "100% (24/24)",
          "n": 24,
          "ci": [
            0.862,
            1
          ],
          "spanKind": "ci95",
          "context": "Claude Code · effort high · eight hard validated tasks"
        },
        {
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-total-latency",
          "series": "Total time per call on hard tasks (separate batches)",
          "point": "Claude Opus 5.5 · Claude Code",
          "metric": "hard-h2h-total-latency",
          "label": "Total time per call on hard tasks (separate batches)",
          "value": 9.18,
          "unit": "seconds",
          "display": "9.18 s",
          "n": 24,
          "range": [
            4.24,
            27.21
          ],
          "spanKind": "minmax",
          "context": "Claude Code · eight hard validated tasks"
        },
        {
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-total-latency",
          "series": "Total time per call on hard tasks (separate batches)",
          "point": "Claude Opus 5.5 (high) · Claude Code",
          "metric": "hard-h2h-total-latency",
          "label": "Total time per call on hard tasks (separate batches)",
          "value": 11.03,
          "unit": "seconds",
          "display": "11.0 s",
          "n": 24,
          "range": [
            3.63,
            63
          ],
          "spanKind": "minmax",
          "context": "Claude Code · effort high · eight hard validated tasks"
        },
        {
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-first-useful-latency",
          "series": "Time to first useful output on hard tasks",
          "point": "Claude Opus 5.5 · Claude Code",
          "metric": "hard-h2h-first-useful-latency",
          "label": "Time to first useful output on hard tasks",
          "value": 6.78,
          "unit": "seconds",
          "display": "6.78 s",
          "n": 24,
          "range": [
            2.39,
            21.77
          ],
          "spanKind": "minmax",
          "context": "Claude Code · eight hard validated tasks"
        },
        {
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-first-useful-latency",
          "series": "Time to first useful output on hard tasks",
          "point": "Claude Opus 5.5 (high) · Claude Code",
          "metric": "hard-h2h-first-useful-latency",
          "label": "Time to first useful output on hard tasks",
          "value": 7.13,
          "unit": "seconds",
          "display": "7.13 s",
          "n": 24,
          "range": [
            2.15,
            56.23
          ],
          "spanKind": "minmax",
          "context": "Claude Code · effort high · eight hard validated tasks"
        },
        {
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-output-tokens",
          "series": "Output tokens",
          "point": "Claude Opus 5.5 · Claude Code",
          "metric": "hard-h2h-output-tokens/Output tokens",
          "label": "Output tokens per call on hard tasks (Output tokens)",
          "value": 945,
          "unit": "tokens",
          "display": "945",
          "n": 24,
          "context": "Claude Code · eight hard validated tasks"
        },
        {
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-output-tokens",
          "series": "Output tokens",
          "point": "Claude Opus 5.5 (high) · Claude Code",
          "metric": "hard-h2h-output-tokens/Output tokens",
          "label": "Output tokens per call on hard tasks (Output tokens)",
          "value": 1052,
          "unit": "tokens",
          "display": "1,052",
          "n": 24,
          "context": "Claude Code · effort high · eight hard validated tasks"
        },
        {
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-cost-per-pass",
          "series": "Cost per strict pass",
          "point": "Claude Opus 5.5 · Claude Code",
          "metric": "hard-h2h-cost-per-pass",
          "label": "List-price cost per strict pass on hard tasks (calculation)",
          "value": 0.02824,
          "unit": "usd",
          "display": "$0.028",
          "n": 24,
          "calculation": true,
          "context": "Claude Code · eight hard validated tasks"
        },
        {
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-cost-per-pass",
          "series": "Cost per strict pass",
          "point": "Claude Opus 5.5 (high) · Claude Code",
          "metric": "hard-h2h-cost-per-pass",
          "label": "List-price cost per strict pass on hard tasks (calculation)",
          "value": 0.03337,
          "unit": "usd",
          "display": "$0.033",
          "n": 24,
          "calculation": true,
          "context": "Claude Code · effort high · eight hard validated tasks"
        },
        {
          "studySlug": "coding-agents-head-to-head",
          "chartId": "coding-agents-pass-rate",
          "series": "Passed every hidden check",
          "point": "Claude Opus 5.5 · Claude Code",
          "metric": "coding-agents-pass-rate",
          "label": "Coding sessions that passed every hidden check",
          "value": 1,
          "unit": "rate",
          "display": "100% (12/12)",
          "n": 12,
          "ci": [
            0.7575,
            1
          ],
          "spanKind": "ci95",
          "context": "Claude Code · six small repository tasks with hidden tests"
        },
        {
          "studySlug": "coding-agents-head-to-head",
          "chartId": "coding-agents-wall-time",
          "series": "Wall time per session",
          "point": "Claude Opus 5.5 · Claude Code",
          "metric": "coding-agents-wall-time",
          "label": "Time per coding session",
          "value": 56.9,
          "unit": "seconds",
          "display": "56.9 s",
          "n": 12,
          "range": [
            29.8,
            185.8
          ],
          "spanKind": "minmax",
          "context": "Claude Code · six small repository tasks with hidden tests"
        },
        {
          "studySlug": "coding-agents-head-to-head",
          "chartId": "coding-agents-tool-calls",
          "series": "Tool calls per session",
          "point": "Claude Opus 5.5 · Claude Code",
          "metric": "coding-agents-tool-calls",
          "label": "Tool calls per coding session",
          "value": 7.5,
          "unit": "calls",
          "display": "7.5",
          "n": 12,
          "range": [
            5,
            14
          ],
          "spanKind": "minmax",
          "context": "Claude Code · six small repository tasks with hidden tests"
        },
        {
          "studySlug": "coding-agents-head-to-head",
          "chartId": "coding-agents-cost-per-pass",
          "series": "List-price cost per pass",
          "point": "Claude Opus 5.5 · Claude Code",
          "metric": "coding-agents-cost-per-pass",
          "label": "List-price cost per passing coding session (calculation)",
          "value": 0.2229,
          "unit": "usd",
          "display": "$0.22",
          "n": 12,
          "calculation": true,
          "context": "Claude Code · six small repository tasks with hidden tests"
        },
        {
          "studySlug": "effort-ladder",
          "chartId": "effort-ladder-pass-rate",
          "series": "Strict pass",
          "point": "Claude Opus 5.5 (low) · Claude Code",
          "metric": "effort-ladder-pass-rate",
          "label": "Strict pass rate by effort on eight hard tasks",
          "value": 1,
          "unit": "rate",
          "display": "100% (16/16)",
          "n": 16,
          "ci": [
            0.8064,
            1
          ],
          "spanKind": "ci95",
          "context": "Claude Code · effort low · eight hard validated tasks, effort ladder"
        },
        {
          "studySlug": "effort-ladder",
          "chartId": "effort-ladder-pass-rate",
          "series": "Strict pass",
          "point": "Claude Opus 5.5 (medium) · Claude Code",
          "metric": "effort-ladder-pass-rate",
          "label": "Strict pass rate by effort on eight hard tasks",
          "value": 1,
          "unit": "rate",
          "display": "100% (16/16)",
          "n": 16,
          "ci": [
            0.8064,
            1
          ],
          "spanKind": "ci95",
          "context": "Claude Code · effort medium · eight hard validated tasks, effort ladder"
        },
        {
          "studySlug": "effort-ladder",
          "chartId": "effort-ladder-pass-rate",
          "series": "Strict pass",
          "point": "Claude Opus 5.5 (high) · Claude Code",
          "metric": "effort-ladder-pass-rate",
          "label": "Strict pass rate by effort on eight hard tasks",
          "value": 1,
          "unit": "rate",
          "display": "100% (16/16)",
          "n": 16,
          "ci": [
            0.8064,
            1
          ],
          "spanKind": "ci95",
          "context": "Claude Code · effort high · eight hard validated tasks, effort ladder"
        },
        {
          "studySlug": "effort-ladder",
          "chartId": "effort-ladder-pass-rate",
          "series": "Strict pass",
          "point": "Claude Opus 5.5 · Claude Code",
          "metric": "effort-ladder-pass-rate",
          "label": "Strict pass rate by effort on eight hard tasks",
          "value": 1,
          "unit": "rate",
          "display": "100% (16/16)",
          "n": 16,
          "ci": [
            0.8064,
            1
          ],
          "spanKind": "ci95",
          "context": "Claude Code · eight hard validated tasks, effort ladder"
        },
        {
          "studySlug": "effort-ladder",
          "chartId": "effort-ladder-total-latency",
          "series": "Total time per call by effort on hard tasks",
          "point": "Claude Opus 5.5 (low) · Claude Code",
          "metric": "effort-ladder-total-latency",
          "label": "Total time per call by effort on hard tasks",
          "value": 7.5,
          "unit": "seconds",
          "display": "7.50 s",
          "n": 16,
          "range": [
            3.34,
            15.82
          ],
          "spanKind": "minmax",
          "context": "Claude Code · effort low · eight hard validated tasks, effort ladder"
        },
        {
          "studySlug": "effort-ladder",
          "chartId": "effort-ladder-total-latency",
          "series": "Total time per call by effort on hard tasks",
          "point": "Claude Opus 5.5 (medium) · Claude Code",
          "metric": "effort-ladder-total-latency",
          "label": "Total time per call by effort on hard tasks",
          "value": 9.72,
          "unit": "seconds",
          "display": "9.72 s",
          "n": 16,
          "range": [
            4.78,
            31.36
          ],
          "spanKind": "minmax",
          "context": "Claude Code · effort medium · eight hard validated tasks, effort ladder"
        },
        {
          "studySlug": "effort-ladder",
          "chartId": "effort-ladder-total-latency",
          "series": "Total time per call by effort on hard tasks",
          "point": "Claude Opus 5.5 (high) · Claude Code",
          "metric": "effort-ladder-total-latency",
          "label": "Total time per call by effort on hard tasks",
          "value": 10.11,
          "unit": "seconds",
          "display": "10.1 s",
          "n": 16,
          "range": [
            3.63,
            63
          ],
          "spanKind": "minmax",
          "context": "Claude Code · effort high · eight hard validated tasks, effort ladder"
        },
        {
          "studySlug": "effort-ladder",
          "chartId": "effort-ladder-total-latency",
          "series": "Total time per call by effort on hard tasks",
          "point": "Claude Opus 5.5 · Claude Code",
          "metric": "effort-ladder-total-latency",
          "label": "Total time per call by effort on hard tasks",
          "value": 9.18,
          "unit": "seconds",
          "display": "9.18 s",
          "n": 16,
          "range": [
            4.24,
            27.21
          ],
          "spanKind": "minmax",
          "context": "Claude Code · eight hard validated tasks, effort ladder"
        },
        {
          "studySlug": "effort-ladder",
          "chartId": "effort-ladder-output-tokens",
          "series": "Output tokens",
          "point": "Claude Opus 5.5 (low) · Claude Code",
          "metric": "effort-ladder-output-tokens/Output tokens",
          "label": "Output tokens per call by effort on hard tasks (Output tokens)",
          "value": 594,
          "unit": "tokens",
          "display": "594",
          "n": 16,
          "context": "Claude Code · effort low · eight hard validated tasks, effort ladder"
        },
        {
          "studySlug": "effort-ladder",
          "chartId": "effort-ladder-output-tokens",
          "series": "Output tokens",
          "point": "Claude Opus 5.5 (medium) · Claude Code",
          "metric": "effort-ladder-output-tokens/Output tokens",
          "label": "Output tokens per call by effort on hard tasks (Output tokens)",
          "value": 853,
          "unit": "tokens",
          "display": "853",
          "n": 16,
          "context": "Claude Code · effort medium · eight hard validated tasks, effort ladder"
        },
        {
          "studySlug": "effort-ladder",
          "chartId": "effort-ladder-output-tokens",
          "series": "Output tokens",
          "point": "Claude Opus 5.5 (high) · Claude Code",
          "metric": "effort-ladder-output-tokens/Output tokens",
          "label": "Output tokens per call by effort on hard tasks (Output tokens)",
          "value": 1052,
          "unit": "tokens",
          "display": "1,052",
          "n": 16,
          "context": "Claude Code · effort high · eight hard validated tasks, effort ladder"
        },
        {
          "studySlug": "effort-ladder",
          "chartId": "effort-ladder-output-tokens",
          "series": "Output tokens",
          "point": "Claude Opus 5.5 · Claude Code",
          "metric": "effort-ladder-output-tokens/Output tokens",
          "label": "Output tokens per call by effort on hard tasks (Output tokens)",
          "value": 945,
          "unit": "tokens",
          "display": "945",
          "n": 16,
          "context": "Claude Code · eight hard validated tasks, effort ladder"
        },
        {
          "studySlug": "effort-ladder",
          "chartId": "effort-ladder-cost-per-pass",
          "series": "Cost per strict pass",
          "point": "Claude Opus 5.5 (low) · Claude Code",
          "metric": "effort-ladder-cost-per-pass",
          "label": "List-price cost per strict pass by effort (calculation)",
          "value": 0.02115,
          "unit": "usd",
          "display": "$0.021",
          "n": 16,
          "calculation": true,
          "context": "Claude Code · effort low · eight hard validated tasks, effort ladder"
        },
        {
          "studySlug": "effort-ladder",
          "chartId": "effort-ladder-cost-per-pass",
          "series": "Cost per strict pass",
          "point": "Claude Opus 5.5 (medium) · Claude Code",
          "metric": "effort-ladder-cost-per-pass",
          "label": "List-price cost per strict pass by effort (calculation)",
          "value": 0.02947,
          "unit": "usd",
          "display": "$0.029",
          "n": 16,
          "calculation": true,
          "context": "Claude Code · effort medium · eight hard validated tasks, effort ladder"
        },
        {
          "studySlug": "effort-ladder",
          "chartId": "effort-ladder-cost-per-pass",
          "series": "Cost per strict pass",
          "point": "Claude Opus 5.5 (high) · Claude Code",
          "metric": "effort-ladder-cost-per-pass",
          "label": "List-price cost per strict pass by effort (calculation)",
          "value": 0.03368,
          "unit": "usd",
          "display": "$0.034",
          "n": 16,
          "calculation": true,
          "context": "Claude Code · effort high · eight hard validated tasks, effort ladder"
        },
        {
          "studySlug": "effort-ladder",
          "chartId": "effort-ladder-cost-per-pass",
          "series": "Cost per strict pass",
          "point": "Claude Opus 5.5 · Claude Code",
          "metric": "effort-ladder-cost-per-pass",
          "label": "List-price cost per strict pass by effort (calculation)",
          "value": 0.02893,
          "unit": "usd",
          "display": "$0.029",
          "n": 16,
          "calculation": true,
          "context": "Claude Code · eight hard validated tasks, effort ladder"
        },
        {
          "studySlug": "caching-consistency",
          "chartId": "caching-cost-with-without",
          "series": "With the cache, as recorded",
          "point": "Claude Opus 5.5 · Claude Code",
          "metric": "caching-cost-with-without/With the cache, as recorded",
          "label": "List-price cost of 5-question sessions with and without the cache (calculation) (With the cache, as recorded)",
          "value": 0.255057,
          "unit": "usd",
          "display": "$0.26",
          "n": 15,
          "calculation": true,
          "context": "Claude Code · calculation: 5-turn cached sessions over a fixed ledger"
        },
        {
          "studySlug": "caching-consistency",
          "chartId": "caching-cost-with-without",
          "series": "Without a cache: every input token at the input price",
          "point": "Claude Opus 5.5 · Claude Code",
          "metric": "caching-cost-with-without/Without a cache: every input token at the input price",
          "label": "List-price cost of 5-question sessions with and without the cache (calculation) (Without a cache: every input token at the input price)",
          "value": 0.544228,
          "unit": "usd",
          "display": "$0.54",
          "n": 15,
          "calculation": true,
          "context": "Claude Code · calculation: 5-turn cached sessions over a fixed ledger"
        },
        {
          "studySlug": "caching-consistency",
          "chartId": "caching-latency-first-vs-later",
          "series": "Turn 1 (writes the ledger to the cache)",
          "point": "Claude Opus 5.5 · Claude Code",
          "metric": "caching-latency-first-vs-later/Turn 1 (writes the ledger to the cache)",
          "label": "Time per turn: first turn vs later turns in a cached session (Turn 1 (writes the ledger to the cache))",
          "value": 1.9,
          "unit": "seconds",
          "display": "1.90 s",
          "n": 3,
          "range": [
            1.78,
            4.36
          ],
          "spanKind": "minmax",
          "context": "Claude Code · 5-turn cached sessions over a fixed ledger"
        },
        {
          "studySlug": "caching-consistency",
          "chartId": "caching-latency-first-vs-later",
          "series": "Turns 2-5 (read the ledger from the cache)",
          "point": "Claude Opus 5.5 · Claude Code",
          "metric": "caching-latency-first-vs-later/Turns 2-5 (read the ledger from the cache)",
          "label": "Time per turn: first turn vs later turns in a cached session (Turns 2-5 (read the ledger from the cache))",
          "value": 2.4,
          "unit": "seconds",
          "display": "2.40 s",
          "n": 12,
          "range": [
            1.63,
            12.67
          ],
          "spanKind": "minmax",
          "context": "Claude Code · 5-turn cached sessions over a fixed ledger"
        },
        {
          "studySlug": "cost-thought-experiments",
          "chartId": "repriced-cost-per-resolved",
          "series": "Repriced cost per resolved instance",
          "point": "Claude Opus 5.5",
          "metric": "repriced-cost-per-resolved",
          "label": "Thought experiment: the same tokens at other list prices",
          "value": 5.753,
          "unit": "usd",
          "display": "$5.75",
          "calculation": true,
          "context": "calculation: Agent’s recorded tokens at this model’s list price"
        },
        {
          "studySlug": "cost-thought-experiments",
          "chartId": "prompt-cache-savings",
          "series": "With caching (as recorded)",
          "point": "Claude Opus 5.5",
          "metric": "prompt-cache-savings/With caching (as recorded)",
          "label": "Thought experiment: what prompt caching saved (With caching (as recorded))",
          "value": 143.83,
          "unit": "usd",
          "display": "$143.83",
          "calculation": true,
          "context": "calculation: Agent’s recorded tokens at this model’s list price"
        },
        {
          "studySlug": "cost-thought-experiments",
          "chartId": "prompt-cache-savings",
          "series": "Without caching",
          "point": "Claude Opus 5.5",
          "metric": "prompt-cache-savings/Without caching",
          "label": "Thought experiment: what prompt caching saved (Without caching)",
          "value": 686.66,
          "unit": "usd",
          "display": "$686.66",
          "calculation": true,
          "context": "calculation: Agent’s recorded tokens at this model’s list price"
        },
        {
          "studySlug": "prompt-cache-break-even",
          "chartId": "cache-break-even-reads",
          "series": "1-hour write (2× input), whole prefix new",
          "point": "Claude Opus 5.5 (cache read $0.2 per M)",
          "metric": "cache-break-even-reads/1-hour write (2× input), whole prefix new",
          "label": "Reuses before a cached prefix costs less, by model and write type (calculation) (1-hour write (2× input), whole prefix new)",
          "value": 1.05,
          "unit": "score",
          "display": "1.05",
          "calculation": true,
          "context": "cache read $0.2 per M · calculation: Agent’s recorded tokens at this model’s list price"
        },
        {
          "studySlug": "prompt-cache-break-even",
          "chartId": "cache-break-even-reads",
          "series": "1-hour write (2× input), whole prefix new",
          "point": "Claude Opus 5.5 (cache read $0.4 per M)",
          "metric": "cache-break-even-reads/1-hour write (2× input), whole prefix new",
          "label": "Reuses before a cached prefix costs less, by model and write type (calculation) (1-hour write (2× input), whole prefix new)",
          "value": 1.11,
          "unit": "score",
          "display": "1.11",
          "calculation": true,
          "context": "cache read $0.4 per M · calculation: Agent’s recorded tokens at this model’s list price"
        },
        {
          "studySlug": "prompt-cache-break-even",
          "chartId": "cache-break-even-reads",
          "series": "1-hour write, pooled n = 6 session share, 19% already cached (as recorded)",
          "point": "Claude Opus 5.5 (cache read $0.2 per M)",
          "metric": "cache-break-even-reads/1-hour write, pooled n = 6 session share, 19% already cached (as recorded)",
          "label": "Reuses before a cached prefix costs less, by model and write type (calculation) (1-hour write, pooled n = 6 session share, 19% already cached (as recorded))",
          "value": 0.67,
          "unit": "score",
          "display": "0.67",
          "calculation": true,
          "context": "cache read $0.2 per M · calculation: Agent’s recorded tokens at this model’s list price"
        },
        {
          "studySlug": "prompt-cache-break-even",
          "chartId": "cache-break-even-reads",
          "series": "1-hour write, pooled n = 6 session share, 19% already cached (as recorded)",
          "point": "Claude Opus 5.5 (cache read $0.4 per M)",
          "metric": "cache-break-even-reads/1-hour write, pooled n = 6 session share, 19% already cached (as recorded)",
          "label": "Reuses before a cached prefix costs less, by model and write type (calculation) (1-hour write, pooled n = 6 session share, 19% already cached (as recorded))",
          "value": 0.72,
          "unit": "score",
          "display": "0.72",
          "calculation": true,
          "context": "cache read $0.4 per M · calculation: Agent’s recorded tokens at this model’s list price"
        },
        {
          "studySlug": "prompt-cache-break-even",
          "chartId": "cache-break-even-reads",
          "series": "5-minute write (1.25× input, an assumption)",
          "point": "Claude Opus 5.5 (cache read $0.2 per M)",
          "metric": "cache-break-even-reads/5-minute write (1.25× input, an assumption)",
          "label": "Reuses before a cached prefix costs less, by model and write type (calculation) (5-minute write (1.25× input, an assumption))",
          "value": 0.26,
          "unit": "score",
          "display": "0.26",
          "calculation": true,
          "context": "cache read $0.2 per M · calculation: Agent’s recorded tokens at this model’s list price"
        },
        {
          "studySlug": "prompt-cache-break-even",
          "chartId": "cache-break-even-reads",
          "series": "5-minute write (1.25× input, an assumption)",
          "point": "Claude Opus 5.5 (cache read $0.4 per M)",
          "metric": "cache-break-even-reads/5-minute write (1.25× input, an assumption)",
          "label": "Reuses before a cached prefix costs less, by model and write type (calculation) (5-minute write (1.25× input, an assumption))",
          "value": 0.28,
          "unit": "score",
          "display": "0.28",
          "calculation": true,
          "context": "cache read $0.4 per M · calculation: Agent’s recorded tokens at this model’s list price"
        },
        {
          "studySlug": "prompt-cache-break-even",
          "chartId": "cache-break-even-cost-curve",
          "series": "Claude Opus 5.5 (no cache)",
          "point": "1 turn",
          "metric": "cache-break-even-cost-curve/1 turn",
          "label": "Cost of a reused prefix with and without the cache, by session length (calculation): 1 turn",
          "value": 31.31,
          "unit": "usd",
          "display": "$31.31",
          "calculation": true,
          "context": "no cache"
        },
        {
          "studySlug": "prompt-cache-break-even",
          "chartId": "cache-break-even-cost-curve",
          "series": "Claude Opus 5.5 (no cache)",
          "point": "2 turns",
          "metric": "cache-break-even-cost-curve/2 turns",
          "label": "Cost of a reused prefix with and without the cache, by session length (calculation): 2 turns",
          "value": 62.62,
          "unit": "usd",
          "display": "$62.62",
          "calculation": true,
          "context": "no cache"
        },
        {
          "studySlug": "prompt-cache-break-even",
          "chartId": "cache-break-even-cost-curve",
          "series": "Claude Opus 5.5 (no cache)",
          "point": "3 turns",
          "metric": "cache-break-even-cost-curve/3 turns",
          "label": "Cost of a reused prefix with and without the cache, by session length (calculation): 3 turns",
          "value": 93.94,
          "unit": "usd",
          "display": "$93.94",
          "calculation": true,
          "context": "no cache"
        },
        {
          "studySlug": "prompt-cache-break-even",
          "chartId": "cache-break-even-cost-curve",
          "series": "Claude Opus 5.5 (no cache)",
          "point": "5 turns",
          "metric": "cache-break-even-cost-curve/5 turns",
          "label": "Cost of a reused prefix with and without the cache, by session length (calculation): 5 turns",
          "value": 156.56,
          "unit": "usd",
          "display": "$156.56",
          "calculation": true,
          "context": "no cache"
        },
        {
          "studySlug": "prompt-cache-break-even",
          "chartId": "cache-break-even-cost-curve",
          "series": "Claude Opus 5.5 (no cache)",
          "point": "10 turns",
          "metric": "cache-break-even-cost-curve/10 turns",
          "label": "Cost of a reused prefix with and without the cache, by session length (calculation): 10 turns",
          "value": 313.12,
          "unit": "usd",
          "display": "$313.12",
          "calculation": true,
          "context": "no cache"
        },
        {
          "studySlug": "prompt-cache-break-even",
          "chartId": "cache-break-even-cost-curve",
          "series": "Claude Opus 5.5 (no cache)",
          "point": "20 turns",
          "metric": "cache-break-even-cost-curve/20 turns",
          "label": "Cost of a reused prefix with and without the cache, by session length (calculation): 20 turns",
          "value": 626.24,
          "unit": "usd",
          "display": "$626.24",
          "calculation": true,
          "context": "no cache"
        },
        {
          "studySlug": "prompt-cache-break-even",
          "chartId": "cache-break-even-cost-curve",
          "series": "Claude Opus 5.5 (1-hour cache write, read $0.2 per M)",
          "point": "1 turn",
          "metric": "cache-break-even-cost-curve/1 turn",
          "label": "Cost of a reused prefix with and without the cache, by session length (calculation): 1 turn",
          "value": 62.62,
          "unit": "usd",
          "display": "$62.62",
          "calculation": true,
          "context": "1-hour cache write · read $0.2 per M"
        },
        {
          "studySlug": "prompt-cache-break-even",
          "chartId": "cache-break-even-cost-curve",
          "series": "Claude Opus 5.5 (1-hour cache write, read $0.2 per M)",
          "point": "2 turns",
          "metric": "cache-break-even-cost-curve/2 turns",
          "label": "Cost of a reused prefix with and without the cache, by session length (calculation): 2 turns",
          "value": 64.19,
          "unit": "usd",
          "display": "$64.19",
          "calculation": true,
          "context": "1-hour cache write · read $0.2 per M"
        },
        {
          "studySlug": "prompt-cache-break-even",
          "chartId": "cache-break-even-cost-curve",
          "series": "Claude Opus 5.5 (1-hour cache write, read $0.2 per M)",
          "point": "3 turns",
          "metric": "cache-break-even-cost-curve/3 turns",
          "label": "Cost of a reused prefix with and without the cache, by session length (calculation): 3 turns",
          "value": 65.76,
          "unit": "usd",
          "display": "$65.76",
          "calculation": true,
          "context": "1-hour cache write · read $0.2 per M"
        },
        {
          "studySlug": "prompt-cache-break-even",
          "chartId": "cache-break-even-cost-curve",
          "series": "Claude Opus 5.5 (1-hour cache write, read $0.2 per M)",
          "point": "5 turns",
          "metric": "cache-break-even-cost-curve/5 turns",
          "label": "Cost of a reused prefix with and without the cache, by session length (calculation): 5 turns",
          "value": 68.89,
          "unit": "usd",
          "display": "$68.89",
          "calculation": true,
          "context": "1-hour cache write · read $0.2 per M"
        },
        {
          "studySlug": "prompt-cache-break-even",
          "chartId": "cache-break-even-cost-curve",
          "series": "Claude Opus 5.5 (1-hour cache write, read $0.2 per M)",
          "point": "10 turns",
          "metric": "cache-break-even-cost-curve/10 turns",
          "label": "Cost of a reused prefix with and without the cache, by session length (calculation): 10 turns",
          "value": 76.71,
          "unit": "usd",
          "display": "$76.71",
          "calculation": true,
          "context": "1-hour cache write · read $0.2 per M"
        },
        {
          "studySlug": "prompt-cache-break-even",
          "chartId": "cache-break-even-cost-curve",
          "series": "Claude Opus 5.5 (1-hour cache write, read $0.2 per M)",
          "point": "20 turns",
          "metric": "cache-break-even-cost-curve/20 turns",
          "label": "Cost of a reused prefix with and without the cache, by session length (calculation): 20 turns",
          "value": 92.37,
          "unit": "usd",
          "display": "$92.37",
          "calculation": true,
          "context": "1-hour cache write · read $0.2 per M"
        },
        {
          "studySlug": "prompt-cache-break-even",
          "chartId": "cache-break-even-cost-curve",
          "series": "Claude Opus 5.5 (1-hour cache write, read $0.4 per M)",
          "point": "1 turn",
          "metric": "cache-break-even-cost-curve/1 turn",
          "label": "Cost of a reused prefix with and without the cache, by session length (calculation): 1 turn",
          "value": 62.62,
          "unit": "usd",
          "display": "$62.62",
          "calculation": true,
          "context": "1-hour cache write · read $0.4 per M"
        },
        {
          "studySlug": "prompt-cache-break-even",
          "chartId": "cache-break-even-cost-curve",
          "series": "Claude Opus 5.5 (1-hour cache write, read $0.4 per M)",
          "point": "2 turns",
          "metric": "cache-break-even-cost-curve/2 turns",
          "label": "Cost of a reused prefix with and without the cache, by session length (calculation): 2 turns",
          "value": 65.76,
          "unit": "usd",
          "display": "$65.76",
          "calculation": true,
          "context": "1-hour cache write · read $0.4 per M"
        },
        {
          "studySlug": "prompt-cache-break-even",
          "chartId": "cache-break-even-cost-curve",
          "series": "Claude Opus 5.5 (1-hour cache write, read $0.4 per M)",
          "point": "3 turns",
          "metric": "cache-break-even-cost-curve/3 turns",
          "label": "Cost of a reused prefix with and without the cache, by session length (calculation): 3 turns",
          "value": 68.89,
          "unit": "usd",
          "display": "$68.89",
          "calculation": true,
          "context": "1-hour cache write · read $0.4 per M"
        },
        {
          "studySlug": "prompt-cache-break-even",
          "chartId": "cache-break-even-cost-curve",
          "series": "Claude Opus 5.5 (1-hour cache write, read $0.4 per M)",
          "point": "5 turns",
          "metric": "cache-break-even-cost-curve/5 turns",
          "label": "Cost of a reused prefix with and without the cache, by session length (calculation): 5 turns",
          "value": 75.15,
          "unit": "usd",
          "display": "$75.15",
          "calculation": true,
          "context": "1-hour cache write · read $0.4 per M"
        },
        {
          "studySlug": "prompt-cache-break-even",
          "chartId": "cache-break-even-cost-curve",
          "series": "Claude Opus 5.5 (1-hour cache write, read $0.4 per M)",
          "point": "10 turns",
          "metric": "cache-break-even-cost-curve/10 turns",
          "label": "Cost of a reused prefix with and without the cache, by session length (calculation): 10 turns",
          "value": 90.8,
          "unit": "usd",
          "display": "$90.80",
          "calculation": true,
          "context": "1-hour cache write · read $0.4 per M"
        },
        {
          "studySlug": "prompt-cache-break-even",
          "chartId": "cache-break-even-cost-curve",
          "series": "Claude Opus 5.5 (1-hour cache write, read $0.4 per M)",
          "point": "20 turns",
          "metric": "cache-break-even-cost-curve/20 turns",
          "label": "Cost of a reused prefix with and without the cache, by session length (calculation): 20 turns",
          "value": 122.12,
          "unit": "usd",
          "display": "$122.12",
          "calculation": true,
          "context": "1-hour cache write · read $0.4 per M"
        },
        {
          "studySlug": "prompt-cache-break-even",
          "chartId": "cache-break-even-session-split",
          "series": "No cache (the same either way)",
          "point": "Claude Opus 5.5 (cache read $0.2 per M)",
          "metric": "cache-break-even-session-split/No cache (the same either way)",
          "label": "One 10-turn session or ten 1-turn sessions: cost with and without the cache (calculation) (No cache (the same either way))",
          "value": 313.12,
          "unit": "usd",
          "display": "$313.12",
          "calculation": true,
          "context": "cache read $0.2 per M"
        },
        {
          "studySlug": "prompt-cache-break-even",
          "chartId": "cache-break-even-session-split",
          "series": "No cache (the same either way)",
          "point": "Claude Opus 5.5 (cache read $0.4 per M)",
          "metric": "cache-break-even-session-split/No cache (the same either way)",
          "label": "One 10-turn session or ten 1-turn sessions: cost with and without the cache (calculation) (No cache (the same either way))",
          "value": 313.12,
          "unit": "usd",
          "display": "$313.12",
          "calculation": true,
          "context": "cache read $0.4 per M"
        },
        {
          "studySlug": "prompt-cache-break-even",
          "chartId": "cache-break-even-session-split",
          "series": "One 10-turn session, 1-hour cache",
          "point": "Claude Opus 5.5 (cache read $0.2 per M)",
          "metric": "cache-break-even-session-split/One 10-turn session, 1-hour cache",
          "label": "One 10-turn session or ten 1-turn sessions: cost with and without the cache (calculation) (One 10-turn session, 1-hour cache)",
          "value": 76.71,
          "unit": "usd",
          "display": "$76.71",
          "calculation": true,
          "context": "cache read $0.2 per M"
        },
        {
          "studySlug": "prompt-cache-break-even",
          "chartId": "cache-break-even-session-split",
          "series": "One 10-turn session, 1-hour cache",
          "point": "Claude Opus 5.5 (cache read $0.4 per M)",
          "metric": "cache-break-even-session-split/One 10-turn session, 1-hour cache",
          "label": "One 10-turn session or ten 1-turn sessions: cost with and without the cache (calculation) (One 10-turn session, 1-hour cache)",
          "value": 90.8,
          "unit": "usd",
          "display": "$90.80",
          "calculation": true,
          "context": "cache read $0.4 per M"
        },
        {
          "studySlug": "prompt-cache-break-even",
          "chartId": "cache-break-even-session-split",
          "series": "Ten 1-turn sessions, 1-hour cache (assumed no reuse of the new prefix)",
          "point": "Claude Opus 5.5 (cache read $0.2 per M)",
          "metric": "cache-break-even-session-split/Ten 1-turn sessions, 1-hour cache (assumed no reuse of the new prefix)",
          "label": "One 10-turn session or ten 1-turn sessions: cost with and without the cache (calculation) (Ten 1-turn sessions, 1-hour cache (assumed no reuse of the new prefix))",
          "value": 626.24,
          "unit": "usd",
          "display": "$626.24",
          "calculation": true,
          "context": "cache read $0.2 per M"
        },
        {
          "studySlug": "prompt-cache-break-even",
          "chartId": "cache-break-even-session-split",
          "series": "Ten 1-turn sessions, 1-hour cache (assumed no reuse of the new prefix)",
          "point": "Claude Opus 5.5 (cache read $0.4 per M)",
          "metric": "cache-break-even-session-split/Ten 1-turn sessions, 1-hour cache (assumed no reuse of the new prefix)",
          "label": "One 10-turn session or ten 1-turn sessions: cost with and without the cache (calculation) (Ten 1-turn sessions, 1-hour cache (assumed no reuse of the new prefix))",
          "value": 626.24,
          "unit": "usd",
          "display": "$626.24",
          "calculation": true,
          "context": "cache read $0.4 per M"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-share",
          "series": "Median call",
          "point": "Claude Opus 5.5 · Claude Code",
          "metric": "thinking-bill-share",
          "label": "Reasoning share of output tokens per call on hard tasks (calculation)",
          "value": 54.79,
          "unit": "percent",
          "display": "54.8%",
          "n": 24,
          "range": [
            29.92,
            95.6
          ],
          "spanKind": "minmax",
          "calculation": true,
          "polarity": "none",
          "context": "Claude Code"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-share",
          "series": "Median call",
          "point": "Claude Opus 5.5 (high) · Claude Code",
          "metric": "thinking-bill-share",
          "label": "Reasoning share of output tokens per call on hard tasks (calculation)",
          "value": 54.43,
          "unit": "percent",
          "display": "54.4%",
          "n": 24,
          "range": [
            36.14,
            96.23
          ],
          "spanKind": "minmax",
          "calculation": true,
          "polarity": "none",
          "context": "Claude Code · effort high"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-cost-per-call",
          "series": "Reasoning (output tokens)",
          "point": "Claude Opus 5.5 (high) · Claude Code",
          "metric": "thinking-bill-cost-per-call/Reasoning (output tokens)",
          "label": "List-price cost per call: reasoning, remaining output and input (calculation) (Reasoning (output tokens))",
          "value": 0.017969,
          "unit": "usd",
          "display": "$0.018",
          "n": 24,
          "calculation": true,
          "polarity": "none",
          "context": "Claude Code · effort high"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-cost-per-call",
          "series": "Reasoning (output tokens)",
          "point": "Claude Opus 5.5 · Claude Code",
          "metric": "thinking-bill-cost-per-call/Reasoning (output tokens)",
          "label": "List-price cost per call: reasoning, remaining output and input (calculation) (Reasoning (output tokens))",
          "value": 0.012528,
          "unit": "usd",
          "display": "$0.013",
          "n": 24,
          "calculation": true,
          "polarity": "none",
          "context": "Claude Code"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-cost-per-call",
          "series": "Remaining output (visible-answer estimate)",
          "point": "Claude Opus 5.5 (high) · Claude Code",
          "metric": "thinking-bill-cost-per-call/Remaining output (visible-answer estimate)",
          "label": "List-price cost per call: reasoning, remaining output and input (calculation) (Remaining output (visible-answer estimate))",
          "value": 0.00799,
          "unit": "usd",
          "display": "$0.0080",
          "n": 24,
          "calculation": true,
          "polarity": "none",
          "context": "Claude Code · effort high"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-cost-per-call",
          "series": "Remaining output (visible-answer estimate)",
          "point": "Claude Opus 5.5 · Claude Code",
          "metric": "thinking-bill-cost-per-call/Remaining output (visible-answer estimate)",
          "label": "List-price cost per call: reasoning, remaining output and input (calculation) (Remaining output (visible-answer estimate))",
          "value": 0.008003,
          "unit": "usd",
          "display": "$0.0080",
          "n": 24,
          "calculation": true,
          "polarity": "none",
          "context": "Claude Code"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-cost-per-call",
          "series": "Input (prompt, cache priced)",
          "point": "Claude Opus 5.5 (high) · Claude Code",
          "metric": "thinking-bill-cost-per-call/Input (prompt, cache priced)",
          "label": "List-price cost per call: reasoning, remaining output and input (calculation) (Input (prompt, cache priced))",
          "value": 0.007407,
          "unit": "usd",
          "display": "$0.0074",
          "n": 24,
          "calculation": true,
          "polarity": "none",
          "context": "Claude Code · effort high"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-cost-per-call",
          "series": "Input (prompt, cache priced)",
          "point": "Claude Opus 5.5 · Claude Code",
          "metric": "thinking-bill-cost-per-call/Input (prompt, cache priced)",
          "label": "List-price cost per call: reasoning, remaining output and input (calculation) (Input (prompt, cache priced))",
          "value": 0.00771,
          "unit": "usd",
          "display": "$0.0077",
          "n": 24,
          "calculation": true,
          "polarity": "none",
          "context": "Claude Code"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-by-effort",
          "series": "Reasoning cost per strict pass",
          "point": "Claude Opus 5.5 (low) · Claude Code",
          "metric": "thinking-bill-by-effort/Reasoning cost per strict pass",
          "label": "Reasoning cost per strict pass by effort, with the total (calculation) (Reasoning cost per strict pass)",
          "value": 0.005031,
          "unit": "usd",
          "display": "$0.0050",
          "n": 16,
          "calculation": true,
          "polarity": "none",
          "context": "Claude Code · effort low"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-by-effort",
          "series": "Reasoning cost per strict pass",
          "point": "Claude Opus 5.5 (medium) · Claude Code",
          "metric": "thinking-bill-by-effort/Reasoning cost per strict pass",
          "label": "Reasoning cost per strict pass by effort, with the total (calculation) (Reasoning cost per strict pass)",
          "value": 0.01344,
          "unit": "usd",
          "display": "$0.013",
          "n": 16,
          "calculation": true,
          "polarity": "none",
          "context": "Claude Code · effort medium"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-by-effort",
          "series": "Reasoning cost per strict pass",
          "point": "Claude Opus 5.5 (high) · Claude Code",
          "metric": "thinking-bill-by-effort/Reasoning cost per strict pass",
          "label": "Reasoning cost per strict pass by effort, with the total (calculation) (Reasoning cost per strict pass)",
          "value": 0.018034,
          "unit": "usd",
          "display": "$0.018",
          "n": 16,
          "calculation": true,
          "polarity": "none",
          "context": "Claude Code · effort high"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-by-effort",
          "series": "Reasoning cost per strict pass",
          "point": "Claude Opus 5.5 · Claude Code",
          "metric": "thinking-bill-by-effort/Reasoning cost per strict pass",
          "label": "Reasoning cost per strict pass by effort, with the total (calculation) (Reasoning cost per strict pass)",
          "value": 0.013104,
          "unit": "usd",
          "display": "$0.013",
          "n": 16,
          "calculation": true,
          "polarity": "none",
          "context": "Claude Code"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-by-effort",
          "series": "Total cost per strict pass",
          "point": "Claude Opus 5.5 (low) · Claude Code",
          "metric": "thinking-bill-by-effort/Total cost per strict pass",
          "label": "Reasoning cost per strict pass by effort, with the total (calculation) (Total cost per strict pass)",
          "value": 0.021152,
          "unit": "usd",
          "display": "$0.021",
          "n": 16,
          "calculation": true,
          "polarity": "none",
          "context": "Claude Code · effort low"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-by-effort",
          "series": "Total cost per strict pass",
          "point": "Claude Opus 5.5 (medium) · Claude Code",
          "metric": "thinking-bill-by-effort/Total cost per strict pass",
          "label": "Reasoning cost per strict pass by effort, with the total (calculation) (Total cost per strict pass)",
          "value": 0.029475,
          "unit": "usd",
          "display": "$0.029",
          "n": 16,
          "calculation": true,
          "polarity": "none",
          "context": "Claude Code · effort medium"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-by-effort",
          "series": "Total cost per strict pass",
          "point": "Claude Opus 5.5 (high) · Claude Code",
          "metric": "thinking-bill-by-effort/Total cost per strict pass",
          "label": "Reasoning cost per strict pass by effort, with the total (calculation) (Total cost per strict pass)",
          "value": 0.033677,
          "unit": "usd",
          "display": "$0.034",
          "n": 16,
          "calculation": true,
          "polarity": "none",
          "context": "Claude Code · effort high"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-by-effort",
          "series": "Total cost per strict pass",
          "point": "Claude Opus 5.5 · Claude Code",
          "metric": "thinking-bill-by-effort/Total cost per strict pass",
          "label": "Reasoning cost per strict pass by effort, with the total (calculation) (Total cost per strict pass)",
          "value": 0.028925,
          "unit": "usd",
          "display": "$0.029",
          "n": 16,
          "calculation": true,
          "polarity": "none",
          "context": "Claude Code"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-short-vs-hard",
          "series": "Eight hard tasks",
          "point": "Claude Opus 5.5 · Claude Code",
          "metric": "thinking-bill-short-vs-hard/Eight hard tasks",
          "label": "Reasoning share on short tasks vs hard tasks (calculation) (Eight hard tasks)",
          "value": 54.79,
          "unit": "percent",
          "display": "54.8%",
          "n": 24,
          "range": [
            29.92,
            95.6
          ],
          "spanKind": "minmax",
          "calculation": true,
          "polarity": "none",
          "context": "Claude Code"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-short-vs-hard",
          "series": "Eight hard tasks",
          "point": "Claude Opus 5.5 (high) · Claude Code",
          "metric": "thinking-bill-short-vs-hard/Eight hard tasks",
          "label": "Reasoning share on short tasks vs hard tasks (calculation) (Eight hard tasks)",
          "value": 54.43,
          "unit": "percent",
          "display": "54.4%",
          "n": 24,
          "range": [
            36.14,
            96.23
          ],
          "spanKind": "minmax",
          "calculation": true,
          "polarity": "none",
          "context": "Claude Code · effort high"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-short-vs-hard",
          "series": "Five short tasks",
          "point": "Claude Opus 5.5 · Claude Code",
          "metric": "thinking-bill-short-vs-hard/Five short tasks",
          "label": "Reasoning share on short tasks vs hard tasks (calculation) (Five short tasks)",
          "value": 0,
          "unit": "percent",
          "display": "0%",
          "n": 15,
          "range": [
            0,
            93.33
          ],
          "spanKind": "minmax",
          "calculation": true,
          "polarity": "none",
          "context": "Claude Code"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-short-vs-hard",
          "series": "Five short tasks",
          "point": "Claude Opus 5.5 (high) · Claude Code",
          "metric": "thinking-bill-short-vs-hard/Five short tasks",
          "label": "Reasoning share on short tasks vs hard tasks (calculation) (Five short tasks)",
          "value": 43.59,
          "unit": "percent",
          "display": "43.6%",
          "n": 15,
          "range": [
            0,
            93.33
          ],
          "spanKind": "minmax",
          "calculation": true,
          "polarity": "none",
          "context": "Claude Code · effort high"
        },
        {
          "studySlug": "llm-speed-anatomy",
          "chartId": "speed-anatomy-first-text",
          "series": "Time to first text",
          "point": "Claude Opus 5.5 · Claude Code",
          "metric": "speed-anatomy-first-text",
          "label": "Time to first text: a 250-line answer, six models",
          "value": 1.97,
          "unit": "seconds",
          "display": "1.97 s",
          "n": 4,
          "range": [
            1.7,
            2.35
          ],
          "spanKind": "minmax",
          "context": "Claude Code"
        },
        {
          "studySlug": "llm-speed-anatomy",
          "chartId": "speed-anatomy-output-speed",
          "series": "Visible tokens per second",
          "point": "Claude Opus 5.5 · Claude Code",
          "metric": "speed-anatomy-output-speed",
          "label": "Output speed after the first text: visible tokens per second (calculation)",
          "value": 155.5,
          "unit": "tokens",
          "display": "156",
          "n": 4,
          "range": [
            154.6,
            156.4
          ],
          "spanKind": "minmax",
          "calculation": true,
          "context": "Claude Code"
        },
        {
          "studySlug": "llm-speed-anatomy",
          "chartId": "speed-anatomy-chars-per-second",
          "series": "Characters per second",
          "point": "Claude Opus 5.5 · Claude Code",
          "metric": "speed-anatomy-chars-per-second",
          "label": "Output speed in characters per second after the first text (calculation)",
          "value": 347,
          "unit": "count",
          "display": "347",
          "n": 4,
          "range": [
            345,
            349
          ],
          "spanKind": "minmax",
          "calculation": true,
          "context": "Claude Code"
        },
        {
          "studySlug": "llm-speed-anatomy",
          "chartId": "speed-anatomy-prompt-size",
          "series": "Claude Opus 5.5 · Claude Code",
          "point": "1k",
          "metric": "speed-anatomy-prompt-size/1k",
          "label": "Time to first text as the prompt grows: 1k",
          "value": 1.51,
          "unit": "seconds",
          "display": "1.51 s",
          "n": 3,
          "range": [
            1.46,
            2.01
          ],
          "spanKind": "minmax",
          "calculation": true,
          "context": "Claude Code"
        },
        {
          "studySlug": "llm-speed-anatomy",
          "chartId": "speed-anatomy-prompt-size",
          "series": "Claude Opus 5.5 · Claude Code",
          "point": "16k",
          "metric": "speed-anatomy-prompt-size/16k",
          "label": "Time to first text as the prompt grows: 16k",
          "value": 1.74,
          "unit": "seconds",
          "display": "1.74 s",
          "n": 3,
          "range": [
            1.7,
            2.97
          ],
          "spanKind": "minmax",
          "calculation": true,
          "context": "Claude Code"
        },
        {
          "studySlug": "llm-speed-anatomy",
          "chartId": "speed-anatomy-prompt-size",
          "series": "Claude Opus 5.5 · Claude Code",
          "point": "64k",
          "metric": "speed-anatomy-prompt-size/64k",
          "label": "Time to first text as the prompt grows: 64k",
          "value": 1.79,
          "unit": "seconds",
          "display": "1.79 s",
          "n": 3,
          "range": [
            1.72,
            3.72
          ],
          "spanKind": "minmax",
          "calculation": true,
          "context": "Claude Code"
        },
        {
          "studySlug": "llm-speed-anatomy",
          "chartId": "speed-anatomy-total-by-size",
          "series": "1k prompt",
          "point": "Claude Opus 5.5 · Claude Code",
          "metric": "speed-anatomy-total-by-size/1k prompt",
          "label": "Total time per call by prompt size (1k prompt)",
          "value": 1.83,
          "unit": "seconds",
          "display": "1.83 s",
          "n": 3,
          "range": [
            1.82,
            2.41
          ],
          "spanKind": "minmax",
          "context": "Claude Code"
        },
        {
          "studySlug": "llm-speed-anatomy",
          "chartId": "speed-anatomy-total-by-size",
          "series": "16k prompt",
          "point": "Claude Opus 5.5 · Claude Code",
          "metric": "speed-anatomy-total-by-size/16k prompt",
          "label": "Total time per call by prompt size (16k prompt)",
          "value": 2.36,
          "unit": "seconds",
          "display": "2.36 s",
          "n": 3,
          "range": [
            2.11,
            3.4
          ],
          "spanKind": "minmax",
          "context": "Claude Code"
        },
        {
          "studySlug": "llm-speed-anatomy",
          "chartId": "speed-anatomy-total-by-size",
          "series": "64k prompt",
          "point": "Claude Opus 5.5 · Claude Code",
          "metric": "speed-anatomy-total-by-size/64k prompt",
          "label": "Total time per call by prompt size (64k prompt)",
          "value": 2.35,
          "unit": "seconds",
          "display": "2.35 s",
          "n": 3,
          "range": [
            2.26,
            4.29
          ],
          "spanKind": "minmax",
          "context": "Claude Code"
        },
        {
          "studySlug": "llm-speed-anatomy",
          "chartId": "speed-anatomy-lookup-correct",
          "series": "Exact answer",
          "point": "Claude Opus 5.5 · Claude Code",
          "metric": "speed-anatomy-lookup-correct",
          "label": "Exact lookup answers at the 1k, 16k and 64k prompt-size targets",
          "value": 0.5556,
          "unit": "rate",
          "display": "56% (5/9)",
          "n": 9,
          "ci": [
            0.2667,
            0.8112
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Code"
        },
        {
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-pass-rate",
          "series": "Strict pass",
          "point": "Claude Opus 5.5 · Claude Code",
          "metric": "harder-h2h-pass-rate/Strict pass",
          "label": "Pass rate on 4 harder tasks (Strict pass)",
          "value": 0.4167,
          "unit": "rate",
          "display": "42% (5/12)",
          "n": 12,
          "ci": [
            0.1933,
            0.6805
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Code"
        },
        {
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-pass-rate",
          "series": "Lenient (format misses counted)",
          "point": "Claude Opus 5.5 · Claude Code",
          "metric": "harder-h2h-pass-rate/Lenient (format misses counted)",
          "label": "Pass rate on 4 harder tasks (Lenient (format misses counted))",
          "value": 0.5,
          "unit": "rate",
          "display": "50% (6/12)",
          "n": 12,
          "ci": [
            0.2538,
            0.7462
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Code"
        },
        {
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-tool-attempts",
          "series": "Tool attempt",
          "point": "Claude Opus 5.5 · Claude Code",
          "metric": "harder-h2h-tool-attempts",
          "label": "Calls that tried a tool although tools were off",
          "value": 0.4167,
          "unit": "rate",
          "display": "42% (5/12)",
          "n": 12,
          "ci": [
            0.1933,
            0.6805
          ],
          "spanKind": "ci95",
          "polarity": "none",
          "context": "Claude Code"
        },
        {
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-pass-by-task",
          "series": "Claude Opus 5.5 · Claude Code",
          "point": "10x10 nonogram",
          "metric": "harder-h2h-pass-by-task/10x10 nonogram",
          "label": "Strict pass rate by task: 10x10 nonogram",
          "value": 1,
          "unit": "rate",
          "display": "100% (3/3)",
          "n": 3,
          "ci": [
            0.4385,
            1
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Code"
        },
        {
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-pass-by-task",
          "series": "Claude Opus 5.5 · Claude Code",
          "point": "Sudoku, 22 givens",
          "metric": "harder-h2h-pass-by-task/Sudoku, 22 givens",
          "label": "Strict pass rate by task: Sudoku, 22 givens",
          "value": 0,
          "unit": "rate",
          "display": "0% (0/3)",
          "n": 3,
          "ci": [
            0,
            0.5615
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Code"
        },
        {
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-pass-by-task",
          "series": "Claude Opus 5.5 · Claude Code",
          "point": "6x6 Skyscrapers",
          "metric": "harder-h2h-pass-by-task/6x6 Skyscrapers",
          "label": "Strict pass rate by task: 6x6 Skyscrapers",
          "value": 0.3333,
          "unit": "rate",
          "display": "33% (1/3)",
          "n": 3,
          "ci": [
            0.0615,
            0.7923
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Code"
        },
        {
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-pass-by-task",
          "series": "Claude Opus 5.5 · Claude Code",
          "point": "Seeded shuffle output",
          "metric": "harder-h2h-pass-by-task/Seeded shuffle output",
          "label": "Strict pass rate by task: Seeded shuffle output",
          "value": 0.3333,
          "unit": "rate",
          "display": "33% (1/3)",
          "n": 3,
          "ci": [
            0.0615,
            0.7923
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Code"
        },
        {
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-total-latency",
          "series": "Total time per call",
          "point": "Claude Opus 5.5 · Claude Code",
          "metric": "harder-h2h-total-latency",
          "label": "Total time per call on harder tasks",
          "value": 80.34,
          "unit": "seconds",
          "display": "80.3 s",
          "n": 9,
          "range": [
            3.82,
            279.5
          ],
          "spanKind": "minmax",
          "context": "Claude Code"
        },
        {
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-output-tokens",
          "series": "Output tokens",
          "point": "Claude Opus 5.5 · Claude Code",
          "metric": "harder-h2h-output-tokens/Output tokens",
          "label": "Output tokens per call on harder tasks (Output tokens)",
          "value": 8420,
          "unit": "tokens",
          "display": "8,420",
          "n": 9,
          "range": [
            279,
            40044
          ],
          "spanKind": "minmax",
          "polarity": "none",
          "context": "Claude Code"
        },
        {
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-cost-per-pass",
          "series": "Cost per strict pass",
          "point": "Claude Opus 5.5 · Claude Code",
          "metric": "harder-h2h-cost-per-pass",
          "label": "List-price cost per strict pass on harder tasks (calculation)",
          "value": 0.59333,
          "unit": "usd",
          "display": "$0.59",
          "n": 12,
          "calculation": true,
          "context": "Claude Code"
        }
      ]
    },
    {
      "slug": "claude-haiku-4-5",
      "name": "Claude Haiku 4.5",
      "vendor": "Anthropic",
      "kind": "model",
      "description": "Anthropic’s small, low-price Claude model, measured through Claude Code, as an LLM router and in the public SWE-bench panel.",
      "aliases": [
        "Claude Haiku 4.5",
        "Haiku 4.5",
        "Claude 4.5 Haiku"
      ],
      "facts": [
        {
          "studySlug": "swe-bench-verified",
          "chartId": "swebench-same-instance-leaderboard",
          "series": "Resolved rate",
          "point": "Claude 4.5 Haiku (high)",
          "metric": "swebench-same-instance-leaderboard",
          "label": "Resolved rate on the same 33 SWE-bench Verified instances",
          "value": 0.7576,
          "unit": "rate",
          "display": "76% (25/33)",
          "n": 33,
          "ci": [
            0.5898,
            0.8717
          ],
          "spanKind": "ci95",
          "context": "effort high · public mini-SWE-agent v2 run, same instances"
        },
        {
          "studySlug": "swe-bench-verified",
          "chartId": "swebench-model-calls",
          "series": "Mean calls",
          "point": "Claude 4.5 Haiku (high)",
          "metric": "swebench-model-calls",
          "label": "Model calls per instance",
          "value": 68.5,
          "unit": "calls",
          "display": "68.5",
          "n": 33,
          "context": "effort high · public mini-SWE-agent v2 run, same instances"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-pass-rate",
          "series": "Pass rate",
          "point": "Claude Haiku 4.5 · Claude Code",
          "metric": "h2h-pass-rate",
          "label": "Pass rate on five validated tasks",
          "value": 1,
          "unit": "rate",
          "display": "100% (15/15)",
          "n": 15,
          "ci": [
            0.7961,
            1
          ],
          "spanKind": "ci95",
          "context": "Claude Code · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-total-latency",
          "series": "Total time per call",
          "point": "Claude Haiku 4.5 · Claude Code",
          "metric": "h2h-total-latency",
          "label": "Total time per call",
          "value": 4.43,
          "unit": "seconds",
          "display": "4.43 s",
          "n": 15,
          "range": [
            3.16,
            23.57
          ],
          "spanKind": "minmax",
          "context": "Claude Code · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-first-useful-latency",
          "series": "Time to first useful output",
          "point": "Claude Haiku 4.5 · Claude Code",
          "metric": "h2h-first-useful-latency",
          "label": "Time to first useful output",
          "value": 3.63,
          "unit": "seconds",
          "display": "3.63 s",
          "n": 15,
          "range": [
            2.78,
            22.27
          ],
          "spanKind": "minmax",
          "context": "Claude Code · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-input-tokens",
          "series": "Cache read",
          "point": "Claude Haiku 4.5 · Claude Code",
          "metric": "h2h-input-tokens/Cache read",
          "label": "Input tokens per call: what the CLI sends (Cache read)",
          "value": 0,
          "unit": "tokens",
          "display": "0",
          "n": 15,
          "context": "Claude Code · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-input-tokens",
          "series": "Other input",
          "point": "Claude Haiku 4.5 · Claude Code",
          "metric": "h2h-input-tokens/Other input",
          "label": "Input tokens per call: what the CLI sends (Other input)",
          "value": 3790,
          "unit": "tokens",
          "display": "3,790",
          "n": 15,
          "context": "Claude Code · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-output-tokens",
          "series": "Output tokens",
          "point": "Claude Haiku 4.5 · Claude Code",
          "metric": "h2h-output-tokens/Output tokens",
          "label": "Output tokens per call (Output tokens)",
          "value": 367,
          "unit": "tokens",
          "display": "367",
          "n": 15,
          "context": "Claude Code · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-list-price-per-call",
          "series": "Cost per call",
          "point": "Claude Haiku 4.5 · Claude Code",
          "metric": "h2h-list-price-per-call",
          "label": "List-price cost per call (calculation)",
          "value": 0.00566,
          "unit": "usd",
          "display": "$0.0057",
          "n": 15,
          "range": [
            0.00513,
            0.01804
          ],
          "spanKind": "minmax",
          "calculation": true,
          "context": "Claude Code · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-cost-per-pass",
          "series": "Cost per pass",
          "point": "Claude Haiku 4.5 · Claude Code",
          "metric": "h2h-cost-per-pass",
          "label": "List-price cost per passing answer (calculation)",
          "value": 0.00836,
          "unit": "usd",
          "display": "$0.0084",
          "n": 15,
          "calculation": true,
          "context": "Claude Code · five short validated tasks"
        },
        {
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-pass-rate",
          "series": "Strict pass",
          "point": "Claude Haiku 4.5 · Claude Code",
          "metric": "hard-h2h-pass-rate/Strict pass",
          "label": "Pass rate on eight hard tasks (Strict pass)",
          "value": 0.4583,
          "unit": "rate",
          "display": "46% (11/24)",
          "n": 24,
          "ci": [
            0.2789,
            0.6493
          ],
          "spanKind": "ci95",
          "context": "Claude Code · eight hard validated tasks"
        },
        {
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-pass-rate",
          "series": "Lenient (format misses counted)",
          "point": "Claude Haiku 4.5 · Claude Code",
          "metric": "hard-h2h-pass-rate/Lenient (format misses counted)",
          "label": "Pass rate on eight hard tasks (Lenient (format misses counted))",
          "value": 0.6667,
          "unit": "rate",
          "display": "67% (16/24)",
          "n": 24,
          "ci": [
            0.4671,
            0.8203
          ],
          "spanKind": "ci95",
          "context": "Claude Code · eight hard validated tasks"
        },
        {
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-total-latency",
          "series": "Total time per call on hard tasks (separate batches)",
          "point": "Claude Haiku 4.5 · Claude Code",
          "metric": "hard-h2h-total-latency",
          "label": "Total time per call on hard tasks (separate batches)",
          "value": 39.01,
          "unit": "seconds",
          "display": "39.0 s",
          "n": 24,
          "range": [
            15.27,
            75.13
          ],
          "spanKind": "minmax",
          "context": "Claude Code · eight hard validated tasks"
        },
        {
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-first-useful-latency",
          "series": "Time to first useful output on hard tasks",
          "point": "Claude Haiku 4.5 · Claude Code",
          "metric": "hard-h2h-first-useful-latency",
          "label": "Time to first useful output on hard tasks",
          "value": 35.54,
          "unit": "seconds",
          "display": "35.5 s",
          "n": 24,
          "range": [
            12.88,
            70.31
          ],
          "spanKind": "minmax",
          "context": "Claude Code · eight hard validated tasks"
        },
        {
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-output-tokens",
          "series": "Output tokens",
          "point": "Claude Haiku 4.5 · Claude Code",
          "metric": "hard-h2h-output-tokens/Output tokens",
          "label": "Output tokens per call on hard tasks (Output tokens)",
          "value": 5064,
          "unit": "tokens",
          "display": "5,064",
          "n": 24,
          "context": "Claude Code · eight hard validated tasks"
        },
        {
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-cost-per-pass",
          "series": "Cost per strict pass",
          "point": "Claude Haiku 4.5 · Claude Code",
          "metric": "hard-h2h-cost-per-pass",
          "label": "List-price cost per strict pass on hard tasks (calculation)",
          "value": 0.0672,
          "unit": "usd",
          "display": "$0.067",
          "n": 24,
          "calculation": true,
          "context": "Claude Code · eight hard validated tasks"
        },
        {
          "studySlug": "caching-consistency",
          "chartId": "consistency-pass-rate",
          "series": "Exact number",
          "point": "Claude Haiku 4.5 · Claude Code",
          "metric": "consistency-pass-rate/Exact number",
          "label": "Same prompt, 10 times: strict pass rate (Exact number)",
          "value": 0,
          "unit": "rate",
          "display": "0% (0/10)",
          "n": 10,
          "ci": [
            0,
            0.2775
          ],
          "spanKind": "ci95",
          "context": "Claude Code · same prompt repeated 10 times"
        },
        {
          "studySlug": "caching-consistency",
          "chartId": "consistency-pass-rate",
          "series": "JSON object",
          "point": "Claude Haiku 4.5 · Claude Code",
          "metric": "consistency-pass-rate/JSON object",
          "label": "Same prompt, 10 times: strict pass rate (JSON object)",
          "value": 0.1,
          "unit": "rate",
          "display": "10% (1/10)",
          "n": 10,
          "ci": [
            0.0179,
            0.4042
          ],
          "spanKind": "ci95",
          "context": "Claude Code · same prompt repeated 10 times"
        },
        {
          "studySlug": "caching-consistency",
          "chartId": "consistency-pass-rate",
          "series": "Code fix",
          "point": "Claude Haiku 4.5 · Claude Code",
          "metric": "consistency-pass-rate/Code fix",
          "label": "Same prompt, 10 times: strict pass rate (Code fix)",
          "value": 1,
          "unit": "rate",
          "display": "100% (10/10)",
          "n": 10,
          "ci": [
            0.7225,
            1
          ],
          "spanKind": "ci95",
          "context": "Claude Code · same prompt repeated 10 times"
        },
        {
          "studySlug": "caching-consistency",
          "chartId": "consistency-distinct-answers",
          "series": "Exact number",
          "point": "Claude Haiku 4.5 · Claude Code",
          "metric": "consistency-distinct-answers/Exact number",
          "label": "Same prompt, 10 times: how many different answers (Exact number)",
          "value": 1,
          "unit": "count",
          "display": "1",
          "n": 10,
          "context": "Claude Code · same prompt repeated 10 times"
        },
        {
          "studySlug": "caching-consistency",
          "chartId": "consistency-distinct-answers",
          "series": "JSON object",
          "point": "Claude Haiku 4.5 · Claude Code",
          "metric": "consistency-distinct-answers/JSON object",
          "label": "Same prompt, 10 times: how many different answers (JSON object)",
          "value": 1,
          "unit": "count",
          "display": "1",
          "n": 10,
          "context": "Claude Code · same prompt repeated 10 times"
        },
        {
          "studySlug": "caching-consistency",
          "chartId": "consistency-distinct-answers",
          "series": "Code fix",
          "point": "Claude Haiku 4.5 · Claude Code",
          "metric": "consistency-distinct-answers/Code fix",
          "label": "Same prompt, 10 times: how many different answers (Code fix)",
          "value": 6,
          "unit": "count",
          "display": "6",
          "n": 10,
          "context": "Claude Code · same prompt repeated 10 times"
        },
        {
          "studySlug": "caching-consistency",
          "chartId": "consistency-latency-spread",
          "series": "Exact number",
          "point": "Claude Haiku 4.5 · Claude Code",
          "metric": "consistency-latency-spread/Exact number",
          "label": "Same prompt, 10 times: time per call (Exact number)",
          "value": 5.06,
          "unit": "seconds",
          "display": "5.06 s",
          "n": 10,
          "range": [
            4.42,
            6.2
          ],
          "spanKind": "minmax",
          "context": "Claude Code · same prompt repeated 10 times"
        },
        {
          "studySlug": "caching-consistency",
          "chartId": "consistency-latency-spread",
          "series": "JSON object",
          "point": "Claude Haiku 4.5 · Claude Code",
          "metric": "consistency-latency-spread/JSON object",
          "label": "Same prompt, 10 times: time per call (JSON object)",
          "value": 7.03,
          "unit": "seconds",
          "display": "7.03 s",
          "n": 10,
          "range": [
            5.28,
            12.27
          ],
          "spanKind": "minmax",
          "context": "Claude Code · same prompt repeated 10 times"
        },
        {
          "studySlug": "caching-consistency",
          "chartId": "consistency-latency-spread",
          "series": "Code fix",
          "point": "Claude Haiku 4.5 · Claude Code",
          "metric": "consistency-latency-spread/Code fix",
          "label": "Same prompt, 10 times: time per call (Code fix)",
          "value": 5.95,
          "unit": "seconds",
          "display": "5.95 s",
          "n": 10,
          "range": [
            4.89,
            7.33
          ],
          "spanKind": "minmax",
          "context": "Claude Code · same prompt repeated 10 times"
        },
        {
          "studySlug": "agent-memory",
          "chartId": "memory-full-pass",
          "series": "Claude Haiku 4.5",
          "point": "No memory",
          "metric": "memory-full-pass/No memory",
          "label": "Full pass rate by kind of memory: No memory",
          "value": 0.2,
          "unit": "rate",
          "display": "20% (2/10)",
          "n": 10,
          "ci": [
            0.0567,
            0.5098
          ],
          "spanKind": "ci95",
          "context": ""
        },
        {
          "studySlug": "agent-memory",
          "chartId": "memory-full-pass",
          "series": "Claude Haiku 4.5",
          "point": "/init CLAUDE.md",
          "metric": "memory-full-pass//init CLAUDE.md",
          "label": "Full pass rate by kind of memory: /init CLAUDE.md",
          "value": 0.2,
          "unit": "rate",
          "display": "20% (2/10)",
          "n": 10,
          "ci": [
            0.0567,
            0.5098
          ],
          "spanKind": "ci95",
          "context": ""
        },
        {
          "studySlug": "agent-memory",
          "chartId": "memory-full-pass",
          "series": "Claude Haiku 4.5",
          "point": "Curated, 11 lines",
          "metric": "memory-full-pass/Curated, 11 lines",
          "label": "Full pass rate by kind of memory: Curated, 11 lines",
          "value": 0.7,
          "unit": "rate",
          "display": "70% (7/10)",
          "n": 10,
          "ci": [
            0.3968,
            0.8922
          ],
          "spanKind": "ci95",
          "context": ""
        },
        {
          "studySlug": "agent-memory",
          "chartId": "memory-full-pass",
          "series": "Claude Haiku 4.5",
          "point": "Raw notes, 60 lines",
          "metric": "memory-full-pass/Raw notes, 60 lines",
          "label": "Full pass rate by kind of memory: Raw notes, 60 lines",
          "value": 0.6,
          "unit": "rate",
          "display": "60% (6/10)",
          "n": 10,
          "ci": [
            0.3127,
            0.8318
          ],
          "spanKind": "ci95",
          "context": ""
        },
        {
          "studySlug": "agent-memory",
          "chartId": "memory-full-pass",
          "series": "Claude Haiku 4.5",
          "point": "Dreamed notes",
          "metric": "memory-full-pass/Dreamed notes",
          "label": "Full pass rate by kind of memory: Dreamed notes",
          "value": 0.7,
          "unit": "rate",
          "display": "70% (7/10)",
          "n": 10,
          "ci": [
            0.3968,
            0.8922
          ],
          "spanKind": "ci95",
          "context": ""
        },
        {
          "studySlug": "agent-memory",
          "chartId": "memory-full-pass",
          "series": "Claude Haiku 4.5",
          "point": "Handbook, 210 lines",
          "metric": "memory-full-pass/Handbook, 210 lines",
          "label": "Full pass rate by kind of memory: Handbook, 210 lines",
          "value": 0.3,
          "unit": "rate",
          "display": "30% (3/10)",
          "n": 10,
          "ci": [
            0.1078,
            0.6032
          ],
          "spanKind": "ci95",
          "context": ""
        },
        {
          "studySlug": "agent-memory",
          "chartId": "memory-full-pass",
          "series": "Claude Haiku 4.5",
          "point": "Stop hook only",
          "metric": "memory-full-pass/Stop hook only",
          "label": "Full pass rate by kind of memory: Stop hook only",
          "value": 0.8,
          "unit": "rate",
          "display": "80% (8/10)",
          "n": 10,
          "ci": [
            0.4902,
            0.9433
          ],
          "spanKind": "ci95",
          "context": ""
        },
        {
          "studySlug": "agent-memory",
          "chartId": "memory-full-pass",
          "series": "Claude Haiku 4.5",
          "point": "Curated + hook",
          "metric": "memory-full-pass/Curated + hook",
          "label": "Full pass rate by kind of memory: Curated + hook",
          "value": 0.9,
          "unit": "rate",
          "display": "90% (9/10)",
          "n": 10,
          "ci": [
            0.5958,
            0.9821
          ],
          "spanKind": "ci95",
          "context": ""
        },
        {
          "studySlug": "agent-memory",
          "chartId": "memory-team-knowledge-by-model",
          "series": "Claude Haiku 4.5",
          "point": "No memory",
          "metric": "memory-team-knowledge-by-model/No memory",
          "label": "Team knowledge followed, Sonnet vs Haiku: No memory",
          "value": 0,
          "unit": "rate",
          "display": "0% (0/10)",
          "n": 10,
          "ci": [
            0,
            0.2775
          ],
          "spanKind": "ci95",
          "context": ""
        },
        {
          "studySlug": "agent-memory",
          "chartId": "memory-team-knowledge-by-model",
          "series": "Claude Haiku 4.5",
          "point": "/init CLAUDE.md",
          "metric": "memory-team-knowledge-by-model//init CLAUDE.md",
          "label": "Team knowledge followed, Sonnet vs Haiku: /init CLAUDE.md",
          "value": 0.1,
          "unit": "rate",
          "display": "10% (1/10)",
          "n": 10,
          "ci": [
            0.0179,
            0.4042
          ],
          "spanKind": "ci95",
          "context": ""
        },
        {
          "studySlug": "agent-memory",
          "chartId": "memory-team-knowledge-by-model",
          "series": "Claude Haiku 4.5",
          "point": "Curated, 11 lines",
          "metric": "memory-team-knowledge-by-model/Curated, 11 lines",
          "label": "Team knowledge followed, Sonnet vs Haiku: Curated, 11 lines",
          "value": 0.8,
          "unit": "rate",
          "display": "80% (8/10)",
          "n": 10,
          "ci": [
            0.4902,
            0.9433
          ],
          "spanKind": "ci95",
          "context": ""
        },
        {
          "studySlug": "agent-memory",
          "chartId": "memory-team-knowledge-by-model",
          "series": "Claude Haiku 4.5",
          "point": "Raw notes, 60 lines",
          "metric": "memory-team-knowledge-by-model/Raw notes, 60 lines",
          "label": "Team knowledge followed, Sonnet vs Haiku: Raw notes, 60 lines",
          "value": 0.6,
          "unit": "rate",
          "display": "60% (6/10)",
          "n": 10,
          "ci": [
            0.3127,
            0.8318
          ],
          "spanKind": "ci95",
          "context": ""
        },
        {
          "studySlug": "agent-memory",
          "chartId": "memory-team-knowledge-by-model",
          "series": "Claude Haiku 4.5",
          "point": "Dreamed notes",
          "metric": "memory-team-knowledge-by-model/Dreamed notes",
          "label": "Team knowledge followed, Sonnet vs Haiku: Dreamed notes",
          "value": 0.8,
          "unit": "rate",
          "display": "80% (8/10)",
          "n": 10,
          "ci": [
            0.4902,
            0.9433
          ],
          "spanKind": "ci95",
          "context": ""
        },
        {
          "studySlug": "agent-memory",
          "chartId": "memory-team-knowledge-by-model",
          "series": "Claude Haiku 4.5",
          "point": "Handbook, 210 lines",
          "metric": "memory-team-knowledge-by-model/Handbook, 210 lines",
          "label": "Team knowledge followed, Sonnet vs Haiku: Handbook, 210 lines",
          "value": 0.3,
          "unit": "rate",
          "display": "30% (3/10)",
          "n": 10,
          "ci": [
            0.1078,
            0.6032
          ],
          "spanKind": "ci95",
          "context": ""
        },
        {
          "studySlug": "agent-memory",
          "chartId": "memory-team-knowledge-by-model",
          "series": "Claude Haiku 4.5",
          "point": "Stop hook only",
          "metric": "memory-team-knowledge-by-model/Stop hook only",
          "label": "Team knowledge followed, Sonnet vs Haiku: Stop hook only",
          "value": 0.8,
          "unit": "rate",
          "display": "80% (8/10)",
          "n": 10,
          "ci": [
            0.4902,
            0.9433
          ],
          "spanKind": "ci95",
          "context": ""
        },
        {
          "studySlug": "agent-memory",
          "chartId": "memory-team-knowledge-by-model",
          "series": "Claude Haiku 4.5",
          "point": "Curated + hook",
          "metric": "memory-team-knowledge-by-model/Curated + hook",
          "label": "Team knowledge followed, Sonnet vs Haiku: Curated + hook",
          "value": 1,
          "unit": "rate",
          "display": "100% (10/10)",
          "n": 10,
          "ci": [
            0.7225,
            1
          ],
          "spanKind": "ci95",
          "context": ""
        },
        {
          "studySlug": "agent-memory",
          "chartId": "memory-broken-test-command",
          "series": "Claude Haiku 4.5",
          "point": "No memory",
          "metric": "memory-broken-test-command/No memory",
          "label": "A stale README command: who still ran it?: No memory",
          "value": 1,
          "unit": "rate",
          "display": "100% (10/10)",
          "n": 10,
          "ci": [
            0.7225,
            1
          ],
          "spanKind": "ci95",
          "polarity": "lower",
          "context": ""
        },
        {
          "studySlug": "agent-memory",
          "chartId": "memory-broken-test-command",
          "series": "Claude Haiku 4.5",
          "point": "/init CLAUDE.md",
          "metric": "memory-broken-test-command//init CLAUDE.md",
          "label": "A stale README command: who still ran it?: /init CLAUDE.md",
          "value": 1,
          "unit": "rate",
          "display": "100% (10/10)",
          "n": 10,
          "ci": [
            0.7225,
            1
          ],
          "spanKind": "ci95",
          "polarity": "lower",
          "context": ""
        },
        {
          "studySlug": "agent-memory",
          "chartId": "memory-broken-test-command",
          "series": "Claude Haiku 4.5",
          "point": "Curated, 11 lines",
          "metric": "memory-broken-test-command/Curated, 11 lines",
          "label": "A stale README command: who still ran it?: Curated, 11 lines",
          "value": 0,
          "unit": "rate",
          "display": "0% (0/10)",
          "n": 10,
          "ci": [
            0,
            0.2775
          ],
          "spanKind": "ci95",
          "polarity": "lower",
          "context": ""
        },
        {
          "studySlug": "agent-memory",
          "chartId": "memory-broken-test-command",
          "series": "Claude Haiku 4.5",
          "point": "Raw notes, 60 lines",
          "metric": "memory-broken-test-command/Raw notes, 60 lines",
          "label": "A stale README command: who still ran it?: Raw notes, 60 lines",
          "value": 1,
          "unit": "rate",
          "display": "100% (10/10)",
          "n": 10,
          "ci": [
            0.7225,
            1
          ],
          "spanKind": "ci95",
          "polarity": "lower",
          "context": ""
        },
        {
          "studySlug": "agent-memory",
          "chartId": "memory-broken-test-command",
          "series": "Claude Haiku 4.5",
          "point": "Dreamed notes",
          "metric": "memory-broken-test-command/Dreamed notes",
          "label": "A stale README command: who still ran it?: Dreamed notes",
          "value": 0.1,
          "unit": "rate",
          "display": "10% (1/10)",
          "n": 10,
          "ci": [
            0.0179,
            0.4042
          ],
          "spanKind": "ci95",
          "polarity": "lower",
          "context": ""
        },
        {
          "studySlug": "agent-memory",
          "chartId": "memory-broken-test-command",
          "series": "Claude Haiku 4.5",
          "point": "Handbook, 210 lines",
          "metric": "memory-broken-test-command/Handbook, 210 lines",
          "label": "A stale README command: who still ran it?: Handbook, 210 lines",
          "value": 0,
          "unit": "rate",
          "display": "0% (0/10)",
          "n": 10,
          "ci": [
            0,
            0.2775
          ],
          "spanKind": "ci95",
          "polarity": "lower",
          "context": ""
        },
        {
          "studySlug": "agent-memory",
          "chartId": "memory-broken-test-command",
          "series": "Claude Haiku 4.5",
          "point": "Stop hook only",
          "metric": "memory-broken-test-command/Stop hook only",
          "label": "A stale README command: who still ran it?: Stop hook only",
          "value": 1,
          "unit": "rate",
          "display": "100% (10/10)",
          "n": 10,
          "ci": [
            0.7225,
            1
          ],
          "spanKind": "ci95",
          "polarity": "lower",
          "context": ""
        },
        {
          "studySlug": "agent-memory",
          "chartId": "memory-broken-test-command",
          "series": "Claude Haiku 4.5",
          "point": "Curated + hook",
          "metric": "memory-broken-test-command/Curated + hook",
          "label": "A stale README command: who still ran it?: Curated + hook",
          "value": 0,
          "unit": "rate",
          "display": "0% (0/10)",
          "n": 10,
          "ci": [
            0,
            0.2775
          ],
          "spanKind": "ci95",
          "polarity": "lower",
          "context": ""
        },
        {
          "studySlug": "agent-memory",
          "chartId": "memory-cost-per-full-pass",
          "series": "Claude Haiku 4.5",
          "point": "No memory",
          "metric": "memory-cost-per-full-pass/No memory",
          "label": "List-price cost per fully correct result (calculation): No memory",
          "value": 0.3786,
          "unit": "usd",
          "display": "$0.38",
          "n": 2,
          "calculation": true,
          "context": ""
        },
        {
          "studySlug": "agent-memory",
          "chartId": "memory-cost-per-full-pass",
          "series": "Claude Haiku 4.5",
          "point": "/init CLAUDE.md",
          "metric": "memory-cost-per-full-pass//init CLAUDE.md",
          "label": "List-price cost per fully correct result (calculation): /init CLAUDE.md",
          "value": 0.4317,
          "unit": "usd",
          "display": "$0.43",
          "n": 2,
          "calculation": true,
          "context": ""
        },
        {
          "studySlug": "agent-memory",
          "chartId": "memory-cost-per-full-pass",
          "series": "Claude Haiku 4.5",
          "point": "Curated, 11 lines",
          "metric": "memory-cost-per-full-pass/Curated, 11 lines",
          "label": "List-price cost per fully correct result (calculation): Curated, 11 lines",
          "value": 0.1095,
          "unit": "usd",
          "display": "$0.11",
          "n": 7,
          "calculation": true,
          "context": ""
        },
        {
          "studySlug": "agent-memory",
          "chartId": "memory-cost-per-full-pass",
          "series": "Claude Haiku 4.5",
          "point": "Raw notes, 60 lines",
          "metric": "memory-cost-per-full-pass/Raw notes, 60 lines",
          "label": "List-price cost per fully correct result (calculation): Raw notes, 60 lines",
          "value": 0.1255,
          "unit": "usd",
          "display": "$0.13",
          "n": 6,
          "calculation": true,
          "context": ""
        },
        {
          "studySlug": "agent-memory",
          "chartId": "memory-cost-per-full-pass",
          "series": "Claude Haiku 4.5",
          "point": "Dreamed notes",
          "metric": "memory-cost-per-full-pass/Dreamed notes",
          "label": "List-price cost per fully correct result (calculation): Dreamed notes",
          "value": 0.1153,
          "unit": "usd",
          "display": "$0.12",
          "n": 7,
          "calculation": true,
          "context": ""
        },
        {
          "studySlug": "agent-memory",
          "chartId": "memory-cost-per-full-pass",
          "series": "Claude Haiku 4.5",
          "point": "Handbook, 210 lines",
          "metric": "memory-cost-per-full-pass/Handbook, 210 lines",
          "label": "List-price cost per fully correct result (calculation): Handbook, 210 lines",
          "value": 0.2615,
          "unit": "usd",
          "display": "$0.26",
          "n": 3,
          "calculation": true,
          "context": ""
        },
        {
          "studySlug": "agent-memory",
          "chartId": "memory-cost-per-full-pass",
          "series": "Claude Haiku 4.5",
          "point": "Stop hook only",
          "metric": "memory-cost-per-full-pass/Stop hook only",
          "label": "List-price cost per fully correct result (calculation): Stop hook only",
          "value": 0.1419,
          "unit": "usd",
          "display": "$0.14",
          "n": 8,
          "calculation": true,
          "context": ""
        },
        {
          "studySlug": "agent-memory",
          "chartId": "memory-cost-per-full-pass",
          "series": "Claude Haiku 4.5",
          "point": "Curated + hook",
          "metric": "memory-cost-per-full-pass/Curated + hook",
          "label": "List-price cost per fully correct result (calculation): Curated + hook",
          "value": 0.096,
          "unit": "usd",
          "display": "$0.096",
          "n": 9,
          "calculation": true,
          "context": ""
        },
        {
          "studySlug": "agent-memory",
          "chartId": "memory-wall-time",
          "series": "Claude Haiku 4.5",
          "point": "No memory",
          "metric": "memory-wall-time/No memory",
          "label": "Time per session: No memory",
          "value": 54.2,
          "unit": "seconds",
          "display": "54.2 s",
          "n": 10,
          "context": ""
        },
        {
          "studySlug": "agent-memory",
          "chartId": "memory-wall-time",
          "series": "Claude Haiku 4.5",
          "point": "/init CLAUDE.md",
          "metric": "memory-wall-time//init CLAUDE.md",
          "label": "Time per session: /init CLAUDE.md",
          "value": 52.9,
          "unit": "seconds",
          "display": "52.9 s",
          "n": 10,
          "context": ""
        },
        {
          "studySlug": "agent-memory",
          "chartId": "memory-wall-time",
          "series": "Claude Haiku 4.5",
          "point": "Curated, 11 lines",
          "metric": "memory-wall-time/Curated, 11 lines",
          "label": "Time per session: Curated, 11 lines",
          "value": 51.7,
          "unit": "seconds",
          "display": "51.7 s",
          "n": 10,
          "context": ""
        },
        {
          "studySlug": "agent-memory",
          "chartId": "memory-wall-time",
          "series": "Claude Haiku 4.5",
          "point": "Raw notes, 60 lines",
          "metric": "memory-wall-time/Raw notes, 60 lines",
          "label": "Time per session: Raw notes, 60 lines",
          "value": 51,
          "unit": "seconds",
          "display": "51.0 s",
          "n": 10,
          "context": ""
        },
        {
          "studySlug": "agent-memory",
          "chartId": "memory-wall-time",
          "series": "Claude Haiku 4.5",
          "point": "Dreamed notes",
          "metric": "memory-wall-time/Dreamed notes",
          "label": "Time per session: Dreamed notes",
          "value": 51.8,
          "unit": "seconds",
          "display": "51.8 s",
          "n": 10,
          "context": ""
        },
        {
          "studySlug": "agent-memory",
          "chartId": "memory-wall-time",
          "series": "Claude Haiku 4.5",
          "point": "Handbook, 210 lines",
          "metric": "memory-wall-time/Handbook, 210 lines",
          "label": "Time per session: Handbook, 210 lines",
          "value": 49.9,
          "unit": "seconds",
          "display": "49.9 s",
          "n": 10,
          "context": ""
        },
        {
          "studySlug": "agent-memory",
          "chartId": "memory-wall-time",
          "series": "Claude Haiku 4.5",
          "point": "Stop hook only",
          "metric": "memory-wall-time/Stop hook only",
          "label": "Time per session: Stop hook only",
          "value": 68.5,
          "unit": "seconds",
          "display": "68.5 s",
          "n": 10,
          "context": ""
        },
        {
          "studySlug": "agent-memory",
          "chartId": "memory-wall-time",
          "series": "Claude Haiku 4.5",
          "point": "Curated + hook",
          "metric": "memory-wall-time/Curated + hook",
          "label": "Time per session: Curated + hook",
          "value": 52.8,
          "unit": "seconds",
          "display": "52.8 s",
          "n": 10,
          "context": ""
        },
        {
          "studySlug": "routing-jev-vs-llm",
          "chartId": "routing-exact-decisions",
          "series": "Exact rate",
          "point": "Claude Haiku 4.5",
          "metric": "routing-exact-decisions",
          "label": "Typed routing decisions answered exactly right",
          "value": 0.8902,
          "unit": "rate",
          "display": "89% (73/82)",
          "n": 82,
          "ci": [
            0.8044,
            0.9412
          ],
          "spanKind": "ci95",
          "context": "typed routing decisions · via Claude Code"
        },
        {
          "studySlug": "routing-jev-vs-llm",
          "chartId": "routing-key-accuracy",
          "series": "Key accuracy",
          "point": "Claude Haiku 4.5",
          "metric": "routing-key-accuracy",
          "label": "Per-question accuracy",
          "value": 0.9433,
          "unit": "rate",
          "display": "94% (183/194)",
          "n": 194,
          "ci": [
            0.9013,
            0.968
          ],
          "spanKind": "ci95",
          "context": "typed routing decisions · via Claude Code"
        },
        {
          "studySlug": "routing-jev-vs-llm",
          "chartId": "routing-exact-by-decision",
          "series": "Claude Haiku 4.5",
          "point": "Failure class",
          "metric": "routing-exact-by-decision/Failure class",
          "label": "Exact rate by decision type: Failure class",
          "value": 0.9444,
          "unit": "rate",
          "display": "94% (17/18)",
          "n": 18,
          "ci": [
            0.7424,
            0.9901
          ],
          "spanKind": "ci95",
          "context": "typed routing decisions · via Claude Code"
        },
        {
          "studySlug": "routing-jev-vs-llm",
          "chartId": "routing-exact-by-decision",
          "series": "Claude Haiku 4.5",
          "point": "Message intent",
          "metric": "routing-exact-by-decision/Message intent",
          "label": "Exact rate by decision type: Message intent",
          "value": 1,
          "unit": "rate",
          "display": "100% (20/20)",
          "n": 20,
          "ci": [
            0.8389,
            1
          ],
          "spanKind": "ci95",
          "context": "typed routing decisions · via Claude Code"
        },
        {
          "studySlug": "routing-jev-vs-llm",
          "chartId": "routing-exact-by-decision",
          "series": "Claude Haiku 4.5",
          "point": "Is it a rule?",
          "metric": "routing-exact-by-decision/Is it a rule?",
          "label": "Exact rate by decision type: Is it a rule?",
          "value": 1,
          "unit": "rate",
          "display": "100% (12/12)",
          "n": 12,
          "ci": [
            0.7575,
            1
          ],
          "spanKind": "ci95",
          "context": "typed routing decisions · via Claude Code"
        },
        {
          "studySlug": "routing-jev-vs-llm",
          "chartId": "routing-exact-by-decision",
          "series": "Claude Haiku 4.5",
          "point": "Context shape",
          "metric": "routing-exact-by-decision/Context shape",
          "label": "Exact rate by decision type: Context shape",
          "value": 0.75,
          "unit": "rate",
          "display": "75% (24/32)",
          "n": 32,
          "ci": [
            0.5789,
            0.8675
          ],
          "spanKind": "ci95",
          "context": "typed routing decisions · via Claude Code"
        },
        {
          "studySlug": "routing-jev-vs-llm",
          "chartId": "routing-cost-per-1000",
          "series": "Cost",
          "point": "Claude Haiku 4.5",
          "metric": "routing-cost-per-1000",
          "label": "Cost per 1,000 routing decisions",
          "value": 8.924,
          "unit": "usd",
          "display": "$8.92",
          "n": 82,
          "calculation": true,
          "context": "typed routing decisions · via Claude Code"
        },
        {
          "studySlug": "routing-jev-vs-llm",
          "chartId": "routing-decision-latency",
          "series": "Wall time (CLI)",
          "point": "Claude Haiku 4.5",
          "metric": "routing-decision-latency/Wall time (CLI)",
          "label": "Time per routing decision (Wall time (CLI))",
          "value": 12674,
          "unit": "ms",
          "display": "12,674 ms",
          "n": 82,
          "range": [
            12674,
            34413
          ],
          "spanKind": "p50-p95",
          "context": "typed routing decisions · via Claude Code"
        },
        {
          "studySlug": "routing-jev-vs-llm",
          "chartId": "routing-decision-latency",
          "series": "Model time (API)",
          "point": "Claude Haiku 4.5",
          "metric": "routing-decision-latency/Model time (API)",
          "label": "Time per routing decision (Model time (API))",
          "value": 10734,
          "unit": "ms",
          "display": "10,734 ms",
          "n": 82,
          "range": [
            10734,
            32072
          ],
          "spanKind": "p50-p95",
          "context": "typed routing decisions · via Claude Code"
        },
        {
          "studySlug": "routing-overhead",
          "chartId": "router-overhead-decision-latency",
          "series": "Decision time",
          "point": "Claude Haiku 4.5 (thinking on, via Claude Code)",
          "metric": "router-overhead-decision-latency",
          "label": "Time to make one routing decision",
          "value": 12543,
          "unit": "ms",
          "display": "12,543 ms",
          "n": 82,
          "range": [
            12543,
            34481
          ],
          "spanKind": "p50-p95",
          "context": "thinking on · via Claude Code · routing overhead per decision"
        },
        {
          "studySlug": "routing-overhead",
          "chartId": "router-overhead-cli-vs-model-time",
          "series": "Model API time",
          "point": "Claude Haiku 4.5 (thinking on, via Claude Code)",
          "metric": "router-overhead-cli-vs-model-time/Model API time",
          "label": "Where an LLM router’s time goes: model vs CLI (Model API time)",
          "value": 10508,
          "unit": "ms",
          "display": "10,508 ms",
          "n": 82,
          "range": [
            10508,
            32132
          ],
          "spanKind": "p50-p95",
          "context": "thinking on · via Claude Code · routing overhead per decision"
        },
        {
          "studySlug": "routing-overhead",
          "chartId": "router-overhead-cli-vs-model-time",
          "series": "CLI and harness time",
          "point": "Claude Haiku 4.5 (thinking on, via Claude Code)",
          "metric": "router-overhead-cli-vs-model-time/CLI and harness time",
          "label": "Where an LLM router’s time goes: model vs CLI (CLI and harness time)",
          "value": 1698,
          "unit": "ms",
          "display": "1,698 ms",
          "n": 82,
          "range": [
            1698,
            2677
          ],
          "spanKind": "p50-p95",
          "context": "thinking on · via Claude Code · routing overhead per decision"
        },
        {
          "studySlug": "routing-overhead",
          "chartId": "router-overhead-completed",
          "series": "Completed",
          "point": "Claude Haiku 4.5 (thinking on, via Claude Code)",
          "metric": "router-overhead-completed",
          "label": "Routing calls that returned a decision",
          "value": 1,
          "unit": "rate",
          "display": "100% (82/82)",
          "n": 82,
          "ci": [
            0.9552,
            1
          ],
          "spanKind": "ci95",
          "context": "thinking on · via Claude Code · routing overhead per decision"
        },
        {
          "studySlug": "routing-overhead",
          "chartId": "router-overhead-cost-list-price",
          "series": "Cost per 1,000 decisions (list price)",
          "point": "Claude Haiku 4.5 (thinking on, via Claude Code)",
          "metric": "router-overhead-cost-list-price",
          "label": "Cost per 1,000 routing decisions for the model routers (calculation)",
          "value": 8.924,
          "unit": "usd",
          "display": "$8.92",
          "n": 82,
          "calculation": true,
          "context": "thinking on · via Claude Code · calculation: Agent’s recorded tokens at this model’s list price · routing overhead per decision"
        },
        {
          "studySlug": "routing-overhead",
          "chartId": "router-overhead-cost-per-1000-tasks",
          "series": "Every model call routed (49.5 per task)",
          "point": "Claude Haiku 4.5 (thinking on, via Claude Code)",
          "metric": "router-overhead-cost-per-1000-tasks/Every model call routed (49.5 per task)",
          "label": "Added routing cost per 1,000 tasks (calculation) (Every model call routed (49.5 per task))",
          "value": 441.74,
          "unit": "usd",
          "display": "$441.74",
          "calculation": true,
          "context": "thinking on · via Claude Code · calculation per 1,000 tasks from recorded decision counts"
        },
        {
          "studySlug": "routing-overhead",
          "chartId": "router-overhead-cost-per-1000-tasks",
          "series": "Only System One decisions (7 per task)",
          "point": "Claude Haiku 4.5 (thinking on, via Claude Code)",
          "metric": "router-overhead-cost-per-1000-tasks/Only System One decisions (7 per task)",
          "label": "Added routing cost per 1,000 tasks (calculation) (Only System One decisions (7 per task))",
          "value": 62.47,
          "unit": "usd",
          "display": "$62.47",
          "calculation": true,
          "context": "thinking on · via Claude Code · calculation per 1,000 tasks from recorded decision counts"
        },
        {
          "studySlug": "routing-overhead",
          "chartId": "router-overhead-delay-per-task",
          "series": "Every model call routed (49.5 per task)",
          "point": "Claude Haiku 4.5 (thinking on, via Claude Code)",
          "metric": "router-overhead-delay-per-task/Every model call routed (49.5 per task)",
          "label": "Added routing delay per task (calculation) (Every model call routed (49.5 per task))",
          "value": 620.8785,
          "unit": "seconds",
          "display": "620.9 s",
          "calculation": true,
          "context": "thinking on · via Claude Code · calculation per task from recorded decision counts, decisions in line"
        },
        {
          "studySlug": "routing-overhead",
          "chartId": "router-overhead-delay-per-task",
          "series": "Only System One decisions (7 per task)",
          "point": "Claude Haiku 4.5 (thinking on, via Claude Code)",
          "metric": "router-overhead-delay-per-task/Only System One decisions (7 per task)",
          "label": "Added routing delay per task (calculation) (Only System One decisions (7 per task))",
          "value": 87.801,
          "unit": "seconds",
          "display": "87.8 s",
          "calculation": true,
          "context": "thinking on · via Claude Code · calculation per task from recorded decision counts, decisions in line"
        },
        {
          "studySlug": "routing-overhead",
          "chartId": "cli-startup-tax",
          "series": "First output event",
          "point": "Claude Code · Claude Haiku 4.5",
          "metric": "cli-startup-tax/First output event",
          "label": "CLI start-up tax on a one-word answer (First output event)",
          "value": 563,
          "unit": "ms",
          "display": "563 ms",
          "n": 5,
          "range": [
            519,
            726
          ],
          "spanKind": "minmax",
          "context": "Claude Code · CLI start-up, one-word prompt, 5 runs"
        },
        {
          "studySlug": "routing-overhead",
          "chartId": "cli-startup-tax",
          "series": "First model output",
          "point": "Claude Code · Claude Haiku 4.5",
          "metric": "cli-startup-tax/First model output",
          "label": "CLI start-up tax on a one-word answer (First model output)",
          "value": 1461,
          "unit": "ms",
          "display": "1,461 ms",
          "n": 5,
          "range": [
            1206,
            2308
          ],
          "spanKind": "minmax",
          "context": "Claude Code · CLI start-up, one-word prompt, 5 runs"
        },
        {
          "studySlug": "routing-overhead",
          "chartId": "cli-startup-tax",
          "series": "Total wall time",
          "point": "Claude Code · Claude Haiku 4.5",
          "metric": "cli-startup-tax/Total wall time",
          "label": "CLI start-up tax on a one-word answer (Total wall time)",
          "value": 2529,
          "unit": "ms",
          "display": "2,529 ms",
          "n": 5,
          "range": [
            2273,
            3382
          ],
          "spanKind": "minmax",
          "context": "Claude Code · CLI start-up, one-word prompt, 5 runs"
        },
        {
          "studySlug": "routing-overhead",
          "chartId": "cli-startup-input-tokens",
          "series": "Input tokens per call",
          "point": "Claude Code · Claude Haiku 4.5",
          "metric": "cli-startup-input-tokens",
          "label": "Input tokens a CLI sends for a one-word answer",
          "value": 6761,
          "unit": "tokens",
          "display": "6,761",
          "n": 5,
          "context": "Claude Code · CLI start-up, one-word prompt, 5 runs"
        },
        {
          "studySlug": "cost-thought-experiments",
          "chartId": "repriced-cost-per-resolved",
          "series": "Repriced cost per resolved instance",
          "point": "Claude Haiku 4.5",
          "metric": "repriced-cost-per-resolved",
          "label": "Thought experiment: the same tokens at other list prices",
          "value": 1.745,
          "unit": "usd",
          "display": "$1.75",
          "calculation": true,
          "context": "calculation: Agent’s recorded tokens at this model’s list price"
        },
        {
          "studySlug": "cost-thought-experiments",
          "chartId": "cost-per-resolved-agent-vs-panel",
          "series": "Cost per resolved instance",
          "point": "Claude 4.5 Haiku (high)",
          "metric": "cost-per-resolved-agent-vs-panel",
          "label": "Recorded cost per resolved instance: Agent vs the public panel",
          "value": 0.479,
          "unit": "usd",
          "display": "$0.48",
          "n": 25,
          "context": "effort high · public mini-SWE-agent v2 run, same instances"
        },
        {
          "studySlug": "cost-thought-experiments",
          "chartId": "prompt-cache-savings",
          "series": "With caching (as recorded)",
          "point": "Claude Haiku 4.5",
          "metric": "prompt-cache-savings/With caching (as recorded)",
          "label": "Thought experiment: what prompt caching saved (With caching (as recorded))",
          "value": 43.61,
          "unit": "usd",
          "display": "$43.61",
          "calculation": true,
          "context": "calculation: Agent’s recorded tokens at this model’s list price"
        },
        {
          "studySlug": "cost-thought-experiments",
          "chartId": "prompt-cache-savings",
          "series": "Without caching",
          "point": "Claude Haiku 4.5",
          "metric": "prompt-cache-savings/Without caching",
          "label": "Thought experiment: what prompt caching saved (Without caching)",
          "value": 171.66,
          "unit": "usd",
          "display": "$171.66",
          "calculation": true,
          "context": "calculation: Agent’s recorded tokens at this model’s list price"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-pass-rate",
          "series": "Strict pass",
          "point": "Claude Haiku 4.5 (single call) · Claude Code",
          "metric": "agent-loop-pass-rate",
          "label": "Strict pass rate: single call vs agent loop on eight hard tasks",
          "value": 0.4583,
          "unit": "rate",
          "display": "46% (11/24)",
          "n": 24,
          "ci": [
            0.2789,
            0.6493
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Code · single call"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-pass-rate",
          "series": "Strict pass",
          "point": "Claude Haiku 4.5 (agent loop) · Claude Code",
          "metric": "agent-loop-pass-rate",
          "label": "Strict pass rate: single call vs agent loop on eight hard tasks",
          "value": 0.5417,
          "unit": "rate",
          "display": "54% (13/24)",
          "n": 24,
          "ci": [
            0.3507,
            0.7211
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Code · agent loop"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "series": "Claude Haiku 4.5 (single call) · Claude Code",
          "point": "Interval merge fix",
          "metric": "agent-loop-by-task/Interval merge fix",
          "label": "Strict passes per task: single call vs agent loop: Interval merge fix",
          "value": 1,
          "unit": "rate",
          "display": "100% (3/3)",
          "n": 3,
          "ci": [
            0.4385,
            1
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Code · single call"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "series": "Claude Haiku 4.5 (single call) · Claude Code",
          "point": "DST day-length fix",
          "metric": "agent-loop-by-task/DST day-length fix",
          "label": "Strict passes per task: single call vs agent loop: DST day-length fix",
          "value": 0.3333,
          "unit": "rate",
          "display": "33% (1/3)",
          "n": 3,
          "ci": [
            0.0615,
            0.7923
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Code · single call"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "series": "Claude Haiku 4.5 (single call) · Claude Code",
          "point": "CSV parser",
          "metric": "agent-loop-by-task/CSV parser",
          "label": "Strict passes per task: single call vs agent loop: CSV parser",
          "value": 0.6667,
          "unit": "rate",
          "display": "67% (2/3)",
          "n": 3,
          "ci": [
            0.2077,
            0.9385
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Code · single call"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "series": "Claude Haiku 4.5 (single call) · Claude Code",
          "point": "Event-loop order",
          "metric": "agent-loop-by-task/Event-loop order",
          "label": "Strict passes per task: single call vs agent loop: Event-loop order",
          "value": 0,
          "unit": "rate",
          "display": "0% (0/3)",
          "n": 3,
          "ci": [
            0,
            0.5615
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Code · single call"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "series": "Claude Haiku 4.5 (single call) · Claude Code",
          "point": "Room schedule",
          "metric": "agent-loop-by-task/Room schedule",
          "label": "Strict passes per task: single call vs agent loop: Room schedule",
          "value": 0,
          "unit": "rate",
          "display": "0% (0/3)",
          "n": 3,
          "ci": [
            0,
            0.5615
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Code · single call"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "series": "Claude Haiku 4.5 (single call) · Claude Code",
          "point": "SemVer regex",
          "metric": "agent-loop-by-task/SemVer regex",
          "label": "Strict passes per task: single call vs agent loop: SemVer regex",
          "value": 1,
          "unit": "rate",
          "display": "100% (3/3)",
          "n": 3,
          "ci": [
            0.4385,
            1
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Code · single call"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "series": "Claude Haiku 4.5 (single call) · Claude Code",
          "point": "Money refactor",
          "metric": "agent-loop-by-task/Money refactor",
          "label": "Strict passes per task: single call vs agent loop: Money refactor",
          "value": 0.6667,
          "unit": "rate",
          "display": "67% (2/3)",
          "n": 3,
          "ci": [
            0.2077,
            0.9385
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Code · single call"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "series": "Claude Haiku 4.5 (single call) · Claude Code",
          "point": "SQL report",
          "metric": "agent-loop-by-task/SQL report",
          "label": "Strict passes per task: single call vs agent loop: SQL report",
          "value": 0,
          "unit": "rate",
          "display": "0% (0/3)",
          "n": 3,
          "ci": [
            0,
            0.5615
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Code · single call"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "series": "Claude Haiku 4.5 (agent loop) · Claude Code",
          "point": "Interval merge fix",
          "metric": "agent-loop-by-task/Interval merge fix",
          "label": "Strict passes per task: single call vs agent loop: Interval merge fix",
          "value": 0.6667,
          "unit": "rate",
          "display": "67% (2/3)",
          "n": 3,
          "ci": [
            0.2077,
            0.9385
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Code · agent loop"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "series": "Claude Haiku 4.5 (agent loop) · Claude Code",
          "point": "DST day-length fix",
          "metric": "agent-loop-by-task/DST day-length fix",
          "label": "Strict passes per task: single call vs agent loop: DST day-length fix",
          "value": 0.6667,
          "unit": "rate",
          "display": "67% (2/3)",
          "n": 3,
          "ci": [
            0.2077,
            0.9385
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Code · agent loop"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "series": "Claude Haiku 4.5 (agent loop) · Claude Code",
          "point": "CSV parser",
          "metric": "agent-loop-by-task/CSV parser",
          "label": "Strict passes per task: single call vs agent loop: CSV parser",
          "value": 0.6667,
          "unit": "rate",
          "display": "67% (2/3)",
          "n": 3,
          "ci": [
            0.2077,
            0.9385
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Code · agent loop"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "series": "Claude Haiku 4.5 (agent loop) · Claude Code",
          "point": "Event-loop order",
          "metric": "agent-loop-by-task/Event-loop order",
          "label": "Strict passes per task: single call vs agent loop: Event-loop order",
          "value": 1,
          "unit": "rate",
          "display": "100% (3/3)",
          "n": 3,
          "ci": [
            0.4385,
            1
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Code · agent loop"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "series": "Claude Haiku 4.5 (agent loop) · Claude Code",
          "point": "Room schedule",
          "metric": "agent-loop-by-task/Room schedule",
          "label": "Strict passes per task: single call vs agent loop: Room schedule",
          "value": 0.6667,
          "unit": "rate",
          "display": "67% (2/3)",
          "n": 3,
          "ci": [
            0.2077,
            0.9385
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Code · agent loop"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "series": "Claude Haiku 4.5 (agent loop) · Claude Code",
          "point": "SemVer regex",
          "metric": "agent-loop-by-task/SemVer regex",
          "label": "Strict passes per task: single call vs agent loop: SemVer regex",
          "value": 0.6667,
          "unit": "rate",
          "display": "67% (2/3)",
          "n": 3,
          "ci": [
            0.2077,
            0.9385
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Code · agent loop"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "series": "Claude Haiku 4.5 (agent loop) · Claude Code",
          "point": "Money refactor",
          "metric": "agent-loop-by-task/Money refactor",
          "label": "Strict passes per task: single call vs agent loop: Money refactor",
          "value": 0,
          "unit": "rate",
          "display": "0% (0/3)",
          "n": 3,
          "ci": [
            0,
            0.5615
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Code · agent loop"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "series": "Claude Haiku 4.5 (agent loop) · Claude Code",
          "point": "SQL report",
          "metric": "agent-loop-by-task/SQL report",
          "label": "Strict passes per task: single call vs agent loop: SQL report",
          "value": 0,
          "unit": "rate",
          "display": "0% (0/3)",
          "n": 3,
          "ci": [
            0,
            0.5615
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Code · agent loop"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-total-time",
          "series": "Total time per attempt",
          "point": "Claude Haiku 4.5 (single call) · Claude Code",
          "metric": "agent-loop-total-time",
          "label": "Total time per attempt: single call vs agent loop",
          "value": 39.01,
          "unit": "seconds",
          "display": "39.0 s",
          "n": 24,
          "range": [
            15.27,
            75.13
          ],
          "spanKind": "minmax",
          "context": "Claude Code · single call"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-total-time",
          "series": "Total time per attempt",
          "point": "Claude Haiku 4.5 (agent loop) · Claude Code",
          "metric": "agent-loop-total-time",
          "label": "Total time per attempt: single call vs agent loop",
          "value": 56.77,
          "unit": "seconds",
          "display": "56.8 s",
          "n": 24,
          "range": [
            24.53,
            223.7
          ],
          "spanKind": "minmax",
          "context": "Claude Code · agent loop"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-tokens",
          "series": "Input tokens (cache reads included)",
          "point": "Claude Haiku 4.5 (single call) · Claude Code",
          "metric": "agent-loop-tokens/Input tokens (cache reads included)",
          "label": "Tokens per attempt: single call vs agent loop (Input tokens (cache reads included))",
          "value": 3941,
          "unit": "tokens",
          "display": "3,941",
          "n": 24,
          "range": [
            3879,
            4221
          ],
          "spanKind": "minmax",
          "context": "Claude Code · single call"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-tokens",
          "series": "Input tokens (cache reads included)",
          "point": "Claude Haiku 4.5 (agent loop) · Claude Code",
          "metric": "agent-loop-tokens/Input tokens (cache reads included)",
          "label": "Tokens per attempt: single call vs agent loop (Input tokens (cache reads included))",
          "value": 71691,
          "unit": "tokens",
          "display": "71,691",
          "n": 24,
          "range": [
            41732,
            516306
          ],
          "spanKind": "minmax",
          "context": "Claude Code · agent loop"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-tokens",
          "series": "Output tokens",
          "point": "Claude Haiku 4.5 (single call) · Claude Code",
          "metric": "agent-loop-tokens/Output tokens",
          "label": "Tokens per attempt: single call vs agent loop (Output tokens)",
          "value": 5064,
          "unit": "tokens",
          "display": "5,064",
          "n": 24,
          "range": [
            1899,
            9321
          ],
          "spanKind": "minmax",
          "context": "Claude Code · single call"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-tokens",
          "series": "Output tokens",
          "point": "Claude Haiku 4.5 (agent loop) · Claude Code",
          "metric": "agent-loop-tokens/Output tokens",
          "label": "Tokens per attempt: single call vs agent loop (Output tokens)",
          "value": 7912,
          "unit": "tokens",
          "display": "7,912",
          "n": 24,
          "range": [
            2541,
            20654
          ],
          "spanKind": "minmax",
          "context": "Claude Code · agent loop"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-tool-calls",
          "series": "Tool calls per attempt",
          "point": "Claude Haiku 4.5 (agent loop) · Claude Code",
          "metric": "agent-loop-tool-calls",
          "label": "Tool calls per agent-loop attempt",
          "value": 3,
          "unit": "count",
          "display": "3",
          "n": 24,
          "range": [
            2,
            18
          ],
          "spanKind": "minmax",
          "context": "Claude Code · agent loop"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-cost-per-pass",
          "series": "Cost per strict pass",
          "point": "Claude Haiku 4.5 (single call) · Claude Code",
          "metric": "agent-loop-cost-per-pass",
          "label": "List-price cost per strict pass: single call vs agent loop (calculation)",
          "value": 0.0672,
          "unit": "usd",
          "display": "$0.067",
          "n": 24,
          "calculation": true,
          "context": "Claude Code · single call"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-cost-per-pass",
          "series": "Cost per strict pass",
          "point": "Claude Haiku 4.5 (agent loop) · Claude Code",
          "metric": "agent-loop-cost-per-pass",
          "label": "List-price cost per strict pass: single call vs agent loop (calculation)",
          "value": 0.14225,
          "unit": "usd",
          "display": "$0.14",
          "n": 24,
          "calculation": true,
          "context": "Claude Code · agent loop"
        },
        {
          "studySlug": "haiku-thinking-on-off",
          "chartId": "haiku-thinking-router-exact",
          "series": "Exact decisions (every scored question right)",
          "point": "Claude Haiku 4.5 (thinking off) · Claude Code",
          "metric": "haiku-thinking-router-exact/Exact decisions (every scored question right)",
          "label": "Haiku thinking study: typed routing decisions answered exactly right (Exact decisions (every scored question right))",
          "value": 0.8659,
          "unit": "rate",
          "display": "87% (71/82)",
          "n": 82,
          "ci": [
            0.7755,
            0.9234
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Code · thinking off · typed routing decisions, thinking on vs off"
        },
        {
          "studySlug": "haiku-thinking-on-off",
          "chartId": "haiku-thinking-router-exact",
          "series": "Exact decisions (every scored question right)",
          "point": "Claude Haiku 4.5 (thinking on) · Claude Code",
          "metric": "haiku-thinking-router-exact/Exact decisions (every scored question right)",
          "label": "Haiku thinking study: typed routing decisions answered exactly right (Exact decisions (every scored question right))",
          "value": 0.8902,
          "unit": "rate",
          "display": "89% (73/82)",
          "n": 82,
          "ci": [
            0.8044,
            0.9412
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Code · thinking on · typed routing decisions, thinking on vs off"
        },
        {
          "studySlug": "haiku-thinking-on-off",
          "chartId": "haiku-thinking-router-exact",
          "series": "Per-question accuracy",
          "point": "Claude Haiku 4.5 (thinking off) · Claude Code",
          "metric": "haiku-thinking-router-exact/Per-question accuracy",
          "label": "Haiku thinking study: typed routing decisions answered exactly right (Per-question accuracy)",
          "value": 0.9124,
          "unit": "rate",
          "display": "91% (177/194)",
          "n": 194,
          "ci": [
            0.8642,
            0.9446
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Code · thinking off · typed routing decisions, thinking on vs off"
        },
        {
          "studySlug": "haiku-thinking-on-off",
          "chartId": "haiku-thinking-router-exact",
          "series": "Per-question accuracy",
          "point": "Claude Haiku 4.5 (thinking on) · Claude Code",
          "metric": "haiku-thinking-router-exact/Per-question accuracy",
          "label": "Haiku thinking study: typed routing decisions answered exactly right (Per-question accuracy)",
          "value": 0.9433,
          "unit": "rate",
          "display": "94% (183/194)",
          "n": 194,
          "ci": [
            0.9013,
            0.968
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Code · thinking on · typed routing decisions, thinking on vs off"
        },
        {
          "studySlug": "haiku-thinking-on-off",
          "chartId": "haiku-thinking-router-latency",
          "series": "Wall time (CLI)",
          "point": "Claude Haiku 4.5 (thinking off) · Claude Code",
          "metric": "haiku-thinking-router-latency/Wall time (CLI)",
          "label": "Haiku thinking study: time per routing decision (Wall time (CLI))",
          "value": 4.66,
          "unit": "seconds",
          "display": "4.66 s",
          "n": 82,
          "range": [
            4.66,
            8.18
          ],
          "spanKind": "p50-p95",
          "context": "Claude Code · thinking off · typed routing decisions, thinking on vs off"
        },
        {
          "studySlug": "haiku-thinking-on-off",
          "chartId": "haiku-thinking-router-latency",
          "series": "Wall time (CLI)",
          "point": "Claude Haiku 4.5 (thinking on) · Claude Code",
          "metric": "haiku-thinking-router-latency/Wall time (CLI)",
          "label": "Haiku thinking study: time per routing decision (Wall time (CLI))",
          "value": 12.54,
          "unit": "seconds",
          "display": "12.5 s",
          "n": 82,
          "range": [
            12.54,
            34.48
          ],
          "spanKind": "p50-p95",
          "context": "Claude Code · thinking on · typed routing decisions, thinking on vs off"
        },
        {
          "studySlug": "haiku-thinking-on-off",
          "chartId": "haiku-thinking-router-latency",
          "series": "Model time (API)",
          "point": "Claude Haiku 4.5 (thinking off) · Claude Code",
          "metric": "haiku-thinking-router-latency/Model time (API)",
          "label": "Haiku thinking study: time per routing decision (Model time (API))",
          "value": 3.79,
          "unit": "seconds",
          "display": "3.79 s",
          "n": 82,
          "range": [
            3.79,
            7.43
          ],
          "spanKind": "p50-p95",
          "context": "Claude Code · thinking off · typed routing decisions, thinking on vs off"
        },
        {
          "studySlug": "haiku-thinking-on-off",
          "chartId": "haiku-thinking-router-latency",
          "series": "Model time (API)",
          "point": "Claude Haiku 4.5 (thinking on) · Claude Code",
          "metric": "haiku-thinking-router-latency/Model time (API)",
          "label": "Haiku thinking study: time per routing decision (Model time (API))",
          "value": 10.51,
          "unit": "seconds",
          "display": "10.5 s",
          "n": 82,
          "range": [
            10.51,
            32.13
          ],
          "spanKind": "p50-p95",
          "context": "Claude Code · thinking on · typed routing decisions, thinking on vs off"
        },
        {
          "studySlug": "haiku-thinking-on-off",
          "chartId": "haiku-thinking-router-tokens",
          "series": "Thinking tokens",
          "point": "Claude Haiku 4.5 (thinking off) · Claude Code",
          "metric": "haiku-thinking-router-tokens/Thinking tokens",
          "label": "Haiku thinking study: thinking and visible output tokens per routing decision (Thinking tokens)",
          "value": 0,
          "unit": "tokens",
          "display": "0",
          "n": 82,
          "context": "Claude Code · thinking off · typed routing decisions, thinking on vs off"
        },
        {
          "studySlug": "haiku-thinking-on-off",
          "chartId": "haiku-thinking-router-tokens",
          "series": "Thinking tokens",
          "point": "Claude Haiku 4.5 (thinking on) · Claude Code",
          "metric": "haiku-thinking-router-tokens/Thinking tokens",
          "label": "Haiku thinking study: thinking and visible output tokens per routing decision (Thinking tokens)",
          "value": 1101,
          "unit": "tokens",
          "display": "1,101",
          "n": 82,
          "context": "Claude Code · thinking on · typed routing decisions, thinking on vs off"
        },
        {
          "studySlug": "haiku-thinking-on-off",
          "chartId": "haiku-thinking-router-tokens",
          "series": "Visible output tokens",
          "point": "Claude Haiku 4.5 (thinking off) · Claude Code",
          "metric": "haiku-thinking-router-tokens/Visible output tokens",
          "label": "Haiku thinking study: thinking and visible output tokens per routing decision (Visible output tokens)",
          "value": 366,
          "unit": "tokens",
          "display": "366",
          "n": 82,
          "context": "Claude Code · thinking off · typed routing decisions, thinking on vs off"
        },
        {
          "studySlug": "haiku-thinking-on-off",
          "chartId": "haiku-thinking-router-tokens",
          "series": "Visible output tokens",
          "point": "Claude Haiku 4.5 (thinking on) · Claude Code",
          "metric": "haiku-thinking-router-tokens/Visible output tokens",
          "label": "Haiku thinking study: thinking and visible output tokens per routing decision (Visible output tokens)",
          "value": 318,
          "unit": "tokens",
          "display": "318",
          "n": 82,
          "context": "Claude Code · thinking on · typed routing decisions, thinking on vs off"
        },
        {
          "studySlug": "haiku-thinking-on-off",
          "chartId": "haiku-thinking-router-cost",
          "series": "Cost per 1,000 decisions",
          "point": "Claude Haiku 4.5 (thinking off) · Claude Code",
          "metric": "haiku-thinking-router-cost",
          "label": "Haiku thinking study: list-price cost per 1,000 routing decisions (calculation)",
          "value": 3.364,
          "unit": "usd",
          "display": "$3.36",
          "n": 82,
          "calculation": true,
          "context": "Claude Code · thinking off · typed routing decisions, thinking on vs off"
        },
        {
          "studySlug": "haiku-thinking-on-off",
          "chartId": "haiku-thinking-router-cost",
          "series": "Cost per 1,000 decisions",
          "point": "Claude Haiku 4.5 (thinking on) · Claude Code",
          "metric": "haiku-thinking-router-cost",
          "label": "Haiku thinking study: list-price cost per 1,000 routing decisions (calculation)",
          "value": 8.924,
          "unit": "usd",
          "display": "$8.92",
          "n": 82,
          "calculation": true,
          "context": "Claude Code · thinking on · typed routing decisions, thinking on vs off"
        },
        {
          "studySlug": "haiku-thinking-on-off",
          "chartId": "haiku-thinking-hard-pass",
          "series": "Strict pass",
          "point": "Claude Haiku 4.5 (thinking off) · Claude Code",
          "metric": "haiku-thinking-hard-pass/Strict pass",
          "label": "Haiku thinking study: pass rate on eight hard tasks (Strict pass)",
          "value": 0.1667,
          "unit": "rate",
          "display": "17% (4/24)",
          "n": 24,
          "ci": [
            0.0668,
            0.3585
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Code · thinking off · eight hard validated tasks, thinking on vs off"
        },
        {
          "studySlug": "haiku-thinking-on-off",
          "chartId": "haiku-thinking-hard-pass",
          "series": "Strict pass",
          "point": "Claude Haiku 4.5 (thinking on) · Claude Code",
          "metric": "haiku-thinking-hard-pass/Strict pass",
          "label": "Haiku thinking study: pass rate on eight hard tasks (Strict pass)",
          "value": 0.4583,
          "unit": "rate",
          "display": "46% (11/24)",
          "n": 24,
          "ci": [
            0.2789,
            0.6493
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Code · thinking on · eight hard validated tasks, thinking on vs off"
        },
        {
          "studySlug": "haiku-thinking-on-off",
          "chartId": "haiku-thinking-hard-pass",
          "series": "Lenient (format misses counted)",
          "point": "Claude Haiku 4.5 (thinking off) · Claude Code",
          "metric": "haiku-thinking-hard-pass/Lenient (format misses counted)",
          "label": "Haiku thinking study: pass rate on eight hard tasks (Lenient (format misses counted))",
          "value": 0.1667,
          "unit": "rate",
          "display": "17% (4/24)",
          "n": 24,
          "ci": [
            0.0668,
            0.3585
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Code · thinking off · eight hard validated tasks, thinking on vs off"
        },
        {
          "studySlug": "haiku-thinking-on-off",
          "chartId": "haiku-thinking-hard-pass",
          "series": "Lenient (format misses counted)",
          "point": "Claude Haiku 4.5 (thinking on) · Claude Code",
          "metric": "haiku-thinking-hard-pass/Lenient (format misses counted)",
          "label": "Haiku thinking study: pass rate on eight hard tasks (Lenient (format misses counted))",
          "value": 0.6667,
          "unit": "rate",
          "display": "67% (16/24)",
          "n": 24,
          "ci": [
            0.4671,
            0.8203
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Code · thinking on · eight hard validated tasks, thinking on vs off"
        },
        {
          "studySlug": "haiku-thinking-on-off",
          "chartId": "haiku-thinking-hard-time",
          "series": "Total time per call",
          "point": "Claude Haiku 4.5 (thinking off) · Claude Code",
          "metric": "haiku-thinking-hard-time",
          "label": "Haiku thinking study: total time per call on hard tasks",
          "value": 2.95,
          "unit": "seconds",
          "display": "2.95 s",
          "n": 24,
          "range": [
            1.7,
            13.01
          ],
          "spanKind": "minmax",
          "context": "Claude Code · thinking off · eight hard validated tasks, thinking on vs off"
        },
        {
          "studySlug": "haiku-thinking-on-off",
          "chartId": "haiku-thinking-hard-time",
          "series": "Total time per call",
          "point": "Claude Haiku 4.5 (thinking on) · Claude Code",
          "metric": "haiku-thinking-hard-time",
          "label": "Haiku thinking study: total time per call on hard tasks",
          "value": 39.01,
          "unit": "seconds",
          "display": "39.0 s",
          "n": 24,
          "range": [
            15.27,
            75.13
          ],
          "spanKind": "minmax",
          "context": "Claude Code · thinking on · eight hard validated tasks, thinking on vs off"
        },
        {
          "studySlug": "haiku-thinking-on-off",
          "statId": "haiku-thinking-hard-reasoning-on",
          "metric": "stat:haiku-thinking-hard-reasoning-on",
          "label": "Claude Haiku 4.5 (thinking on): median reasoning tokens per hard-task call",
          "value": 4556,
          "unit": "tokens",
          "display": "4,556",
          "n": 24,
          "context": "thinking on"
        },
        {
          "studySlug": "haiku-thinking-on-off",
          "statId": "haiku-thinking-hard-cost-per-pass-off",
          "metric": "stat:haiku-thinking-hard-cost-per-pass-off",
          "label": "Claude Haiku 4.5 (thinking off): list-price cost per strict pass on hard tasks (calculation)",
          "value": 0.03654,
          "unit": "usd",
          "display": "$0.0365",
          "n": 24,
          "calculation": true,
          "context": "thinking off"
        },
        {
          "studySlug": "haiku-thinking-on-off",
          "statId": "haiku-thinking-hard-cost-per-pass-on",
          "metric": "stat:haiku-thinking-hard-cost-per-pass-on",
          "label": "Claude Haiku 4.5 (thinking on): list-price cost per strict pass on hard tasks (calculation)",
          "value": 0.0672,
          "unit": "usd",
          "display": "$0.0672",
          "n": 24,
          "calculation": true,
          "context": "thinking on"
        },
        {
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-pass-rate",
          "series": "Strict pass: the whole reply is the right JSON",
          "point": "Claude Haiku 4.5 (instructions) · Claude Code",
          "metric": "structured-output-pass-rate/Strict pass: the whole reply is the right JSON",
          "label": "Does a JSON schema raise the pass rate? Instructions vs schema mode (Strict pass: the whole reply is the right JSON)",
          "value": 0,
          "unit": "rate",
          "display": "0% (0/24)",
          "n": 24,
          "ci": [
            0,
            0.138
          ],
          "spanKind": "ci95",
          "calculation": true,
          "polarity": "higher",
          "context": "Claude Code · instructions"
        },
        {
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-pass-rate",
          "series": "Strict pass: the whole reply is the right JSON",
          "point": "Claude Haiku 4.5 (JSON schema) · Claude Code",
          "metric": "structured-output-pass-rate/Strict pass: the whole reply is the right JSON",
          "label": "Does a JSON schema raise the pass rate? Instructions vs schema mode (Strict pass: the whole reply is the right JSON)",
          "value": 0.75,
          "unit": "rate",
          "display": "75% (18/24)",
          "n": 24,
          "ci": [
            0.551,
            0.88
          ],
          "spanKind": "ci95",
          "calculation": true,
          "polarity": "higher",
          "context": "Claude Code · JSON schema"
        },
        {
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-pass-rate",
          "series": "Right answer in any format (strict pass or format miss)",
          "point": "Claude Haiku 4.5 (instructions) · Claude Code",
          "metric": "structured-output-pass-rate/Right answer in any format (strict pass or format miss)",
          "label": "Does a JSON schema raise the pass rate? Instructions vs schema mode (Right answer in any format (strict pass or format miss))",
          "value": 0.7083,
          "unit": "rate",
          "display": "71% (17/24)",
          "n": 24,
          "ci": [
            0.5083,
            0.8509
          ],
          "spanKind": "ci95",
          "calculation": true,
          "polarity": "higher",
          "context": "Claude Code · instructions"
        },
        {
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-pass-rate",
          "series": "Right answer in any format (strict pass or format miss)",
          "point": "Claude Haiku 4.5 (JSON schema) · Claude Code",
          "metric": "structured-output-pass-rate/Right answer in any format (strict pass or format miss)",
          "label": "Does a JSON schema raise the pass rate? Instructions vs schema mode (Right answer in any format (strict pass or format miss))",
          "value": 0.75,
          "unit": "rate",
          "display": "75% (18/24)",
          "n": 24,
          "ci": [
            0.551,
            0.88
          ],
          "spanKind": "ci95",
          "calculation": true,
          "polarity": "higher",
          "context": "Claude Code · JSON schema"
        },
        {
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-outcomes",
          "series": "Strict pass",
          "point": "Claude Haiku 4.5 (instructions) · Claude Code",
          "metric": "structured-output-outcomes/Strict pass",
          "label": "What each call produced: strict pass, format miss, wrong values or error (Strict pass)",
          "value": 0,
          "unit": "count",
          "display": "0",
          "n": 24,
          "context": "Claude Code · instructions"
        },
        {
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-outcomes",
          "series": "Strict pass",
          "point": "Claude Haiku 4.5 (JSON schema) · Claude Code",
          "metric": "structured-output-outcomes/Strict pass",
          "label": "What each call produced: strict pass, format miss, wrong values or error (Strict pass)",
          "value": 18,
          "unit": "count",
          "display": "18",
          "n": 24,
          "context": "Claude Code · JSON schema"
        },
        {
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-outcomes",
          "series": "Format miss",
          "point": "Claude Haiku 4.5 (instructions) · Claude Code",
          "metric": "structured-output-outcomes/Format miss",
          "label": "What each call produced: strict pass, format miss, wrong values or error (Format miss)",
          "value": 17,
          "unit": "count",
          "display": "17",
          "n": 24,
          "context": "Claude Code · instructions"
        },
        {
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-outcomes",
          "series": "Format miss",
          "point": "Claude Haiku 4.5 (JSON schema) · Claude Code",
          "metric": "structured-output-outcomes/Format miss",
          "label": "What each call produced: strict pass, format miss, wrong values or error (Format miss)",
          "value": 0,
          "unit": "count",
          "display": "0",
          "n": 24,
          "context": "Claude Code · JSON schema"
        },
        {
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-outcomes",
          "series": "Wrong values",
          "point": "Claude Haiku 4.5 (instructions) · Claude Code",
          "metric": "structured-output-outcomes/Wrong values",
          "label": "What each call produced: strict pass, format miss, wrong values or error (Wrong values)",
          "value": 7,
          "unit": "count",
          "display": "7",
          "n": 24,
          "context": "Claude Code · instructions"
        },
        {
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-outcomes",
          "series": "Wrong values",
          "point": "Claude Haiku 4.5 (JSON schema) · Claude Code",
          "metric": "structured-output-outcomes/Wrong values",
          "label": "What each call produced: strict pass, format miss, wrong values or error (Wrong values)",
          "value": 6,
          "unit": "count",
          "display": "6",
          "n": 24,
          "context": "Claude Code · JSON schema"
        },
        {
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-outcomes",
          "series": "Error",
          "point": "Claude Haiku 4.5 (instructions) · Claude Code",
          "metric": "structured-output-outcomes/Error",
          "label": "What each call produced: strict pass, format miss, wrong values or error (Error)",
          "value": 0,
          "unit": "count",
          "display": "0",
          "n": 24,
          "context": "Claude Code · instructions"
        },
        {
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-outcomes",
          "series": "Error",
          "point": "Claude Haiku 4.5 (JSON schema) · Claude Code",
          "metric": "structured-output-outcomes/Error",
          "label": "What each call produced: strict pass, format miss, wrong values or error (Error)",
          "value": 0,
          "unit": "count",
          "display": "0",
          "n": 24,
          "context": "Claude Code · JSON schema"
        },
        {
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-time",
          "series": "Median time per call (the three prompts pooled)",
          "point": "Claude Haiku 4.5 (instructions) · Claude Code",
          "metric": "structured-output-time",
          "label": "Time per call, instructions vs schema mode",
          "value": 9.52,
          "unit": "seconds",
          "display": "9.52 s",
          "n": 24,
          "range": [
            5.67,
            17
          ],
          "spanKind": "minmax",
          "context": "Claude Code · instructions"
        },
        {
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-time",
          "series": "Median time per call (the three prompts pooled)",
          "point": "Claude Haiku 4.5 (JSON schema) · Claude Code",
          "metric": "structured-output-time",
          "label": "Time per call, instructions vs schema mode",
          "value": 8.46,
          "unit": "seconds",
          "display": "8.46 s",
          "n": 24,
          "range": [
            5.9,
            12.23
          ],
          "spanKind": "minmax",
          "context": "Claude Code · JSON schema"
        },
        {
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-tokens",
          "series": "Median output tokens per call",
          "point": "Claude Haiku 4.5 (instructions) · Claude Code",
          "metric": "structured-output-tokens/Median output tokens per call",
          "label": "Output and reasoning tokens per call, instructions vs schema mode (Median output tokens per call)",
          "value": 1128,
          "unit": "tokens",
          "display": "1,128",
          "n": 24,
          "context": "Claude Code · instructions"
        },
        {
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-tokens",
          "series": "Median output tokens per call",
          "point": "Claude Haiku 4.5 (JSON schema) · Claude Code",
          "metric": "structured-output-tokens/Median output tokens per call",
          "label": "Output and reasoning tokens per call, instructions vs schema mode (Median output tokens per call)",
          "value": 1036,
          "unit": "tokens",
          "display": "1,036",
          "n": 24,
          "context": "Claude Code · JSON schema"
        },
        {
          "studySlug": "prompt-cache-break-even",
          "chartId": "cache-break-even-reads",
          "series": "1-hour write (2× input), whole prefix new",
          "point": "Claude Haiku 4.5",
          "metric": "cache-break-even-reads/1-hour write (2× input), whole prefix new",
          "label": "Reuses before a cached prefix costs less, by model and write type (calculation) (1-hour write (2× input), whole prefix new)",
          "value": 1.11,
          "unit": "score",
          "display": "1.11",
          "calculation": true,
          "context": "calculation: Agent’s recorded tokens at this model’s list price"
        },
        {
          "studySlug": "prompt-cache-break-even",
          "chartId": "cache-break-even-reads",
          "series": "1-hour write, pooled n = 6 session share, 19% already cached (as recorded)",
          "point": "Claude Haiku 4.5",
          "metric": "cache-break-even-reads/1-hour write, pooled n = 6 session share, 19% already cached (as recorded)",
          "label": "Reuses before a cached prefix costs less, by model and write type (calculation) (1-hour write, pooled n = 6 session share, 19% already cached (as recorded))",
          "value": 0.72,
          "unit": "score",
          "display": "0.72",
          "calculation": true,
          "context": "calculation: Agent’s recorded tokens at this model’s list price"
        },
        {
          "studySlug": "prompt-cache-break-even",
          "chartId": "cache-break-even-reads",
          "series": "5-minute write (1.25× input, an assumption)",
          "point": "Claude Haiku 4.5",
          "metric": "cache-break-even-reads/5-minute write (1.25× input, an assumption)",
          "label": "Reuses before a cached prefix costs less, by model and write type (calculation) (5-minute write (1.25× input, an assumption))",
          "value": 0.28,
          "unit": "score",
          "display": "0.28",
          "calculation": true,
          "context": "calculation: Agent’s recorded tokens at this model’s list price"
        },
        {
          "studySlug": "routing-holdout",
          "chartId": "routing-holdout-exact",
          "series": "Exact rate",
          "point": "Claude Haiku 4.5 · Claude Code",
          "metric": "routing-holdout-exact",
          "label": "Unseen routing decisions answered exactly right",
          "value": 0.7857,
          "unit": "rate",
          "display": "79% (44/56)",
          "n": 56,
          "ci": [
            0.6618,
            0.8729
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Code"
        },
        {
          "studySlug": "routing-holdout",
          "chartId": "routing-holdout-key-accuracy",
          "series": "Key accuracy",
          "point": "Claude Haiku 4.5 · Claude Code",
          "metric": "routing-holdout-key-accuracy",
          "label": "Per-question accuracy on unseen decisions",
          "value": 0.816,
          "unit": "rate",
          "display": "82% (102/125)",
          "n": 125,
          "ci": [
            0.739,
            0.8741
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Code"
        },
        {
          "studySlug": "routing-holdout",
          "chartId": "routing-holdout-by-purpose",
          "series": "Claude Haiku 4.5 · Claude Code",
          "point": "Failure class",
          "metric": "routing-holdout-by-purpose/Failure class",
          "label": "Exact rate on unseen decisions, by decision type: Failure class",
          "value": 0.9286,
          "unit": "rate",
          "display": "93% (13/14)",
          "n": 14,
          "ci": [
            0.6853,
            0.9873
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Code"
        },
        {
          "studySlug": "routing-holdout",
          "chartId": "routing-holdout-by-purpose",
          "series": "Claude Haiku 4.5 · Claude Code",
          "point": "Message intent",
          "metric": "routing-holdout-by-purpose/Message intent",
          "label": "Exact rate on unseen decisions, by decision type: Message intent",
          "value": 0.9286,
          "unit": "rate",
          "display": "93% (13/14)",
          "n": 14,
          "ci": [
            0.6853,
            0.9873
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Code"
        },
        {
          "studySlug": "routing-holdout",
          "chartId": "routing-holdout-by-purpose",
          "series": "Claude Haiku 4.5 · Claude Code",
          "point": "Is it a rule?",
          "metric": "routing-holdout-by-purpose/Is it a rule?",
          "label": "Exact rate on unseen decisions, by decision type: Is it a rule?",
          "value": 0.9286,
          "unit": "rate",
          "display": "93% (13/14)",
          "n": 14,
          "ci": [
            0.6853,
            0.9873
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Code"
        },
        {
          "studySlug": "routing-holdout",
          "chartId": "routing-holdout-by-purpose",
          "series": "Claude Haiku 4.5 · Claude Code",
          "point": "Context shape",
          "metric": "routing-holdout-by-purpose/Context shape",
          "label": "Exact rate on unseen decisions, by decision type: Context shape",
          "value": 0.3571,
          "unit": "rate",
          "display": "36% (5/14)",
          "n": 14,
          "ci": [
            0.1634,
            0.6124
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Code"
        },
        {
          "studySlug": "routing-holdout",
          "chartId": "routing-holdout-tuned-vs-unseen",
          "series": "Tuned set (routing-jev-vs-llm)",
          "point": "Claude Haiku 4.5 · Claude Code",
          "metric": "routing-holdout-tuned-vs-unseen/Tuned set (routing-jev-vs-llm)",
          "label": "Tuned case set vs unseen holdout: exact rate per router (Tuned set (routing-jev-vs-llm))",
          "value": 0.8902,
          "unit": "rate",
          "display": "89% (73/82)",
          "n": 82,
          "ci": [
            0.8044,
            0.9412
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Code"
        },
        {
          "studySlug": "routing-holdout",
          "chartId": "routing-holdout-tuned-vs-unseen",
          "series": "Unseen holdout",
          "point": "Claude Haiku 4.5 · Claude Code",
          "metric": "routing-holdout-tuned-vs-unseen/Unseen holdout",
          "label": "Tuned case set vs unseen holdout: exact rate per router (Unseen holdout)",
          "value": 0.7857,
          "unit": "rate",
          "display": "79% (44/56)",
          "n": 56,
          "ci": [
            0.6618,
            0.8729
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Code"
        },
        {
          "studySlug": "routing-holdout",
          "chartId": "routing-holdout-latency",
          "series": "Wall time",
          "point": "Claude Haiku 4.5 · Claude Code",
          "metric": "routing-holdout-latency/Wall time",
          "label": "Time per routing decision, by route (Wall time)",
          "value": 9.444,
          "unit": "seconds",
          "display": "9.44 s",
          "n": 56,
          "range": [
            9.444,
            25.413
          ],
          "spanKind": "p50-p95",
          "context": "Claude Code"
        },
        {
          "studySlug": "routing-holdout",
          "chartId": "routing-holdout-latency",
          "series": "Model time (API, CLI-reported)",
          "point": "Claude Haiku 4.5 · Claude Code",
          "metric": "routing-holdout-latency/Model time (API, CLI-reported)",
          "label": "Time per routing decision, by route (Model time (API, CLI-reported))",
          "value": 7.522,
          "unit": "seconds",
          "display": "7.52 s",
          "n": 56,
          "range": [
            7.522,
            23.913
          ],
          "spanKind": "p50-p95",
          "context": "Claude Code"
        },
        {
          "studySlug": "routing-holdout",
          "chartId": "routing-holdout-cost-per-1000",
          "series": "Cost",
          "point": "Claude Haiku 4.5 · Claude Code",
          "metric": "routing-holdout-cost-per-1000",
          "label": "Cost per 1,000 unseen routing decisions",
          "value": 7.129,
          "unit": "usd",
          "display": "$7.13",
          "n": 56,
          "calculation": true,
          "context": "Claude Code"
        },
        {
          "studySlug": "routing-holdout",
          "statId": "holdout-gap-claude-haiku",
          "metric": "stat:holdout-gap-claude-haiku",
          "label": "Claude Haiku 4.5 · Claude Code: holdout minus tuned-set exact rate",
          "value": -0.1045,
          "unit": "rate",
          "display": "−10.5 points",
          "n": 56,
          "calculation": true,
          "context": ""
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-share",
          "series": "Median call",
          "point": "Claude Haiku 4.5 · Claude Code",
          "metric": "thinking-bill-share",
          "label": "Reasoning share of output tokens per call on hard tasks (calculation)",
          "value": 91.68,
          "unit": "percent",
          "display": "91.7%",
          "n": 24,
          "range": [
            76.46,
            99.27
          ],
          "spanKind": "minmax",
          "calculation": true,
          "polarity": "none",
          "context": "Claude Code"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-cost-per-call",
          "series": "Reasoning (output tokens)",
          "point": "Claude Haiku 4.5 · Claude Code",
          "metric": "thinking-bill-cost-per-call/Reasoning (output tokens)",
          "label": "List-price cost per call: reasoning, remaining output and input (calculation) (Reasoning (output tokens))",
          "value": 0.024492,
          "unit": "usd",
          "display": "$0.024",
          "n": 24,
          "calculation": true,
          "polarity": "none",
          "context": "Claude Code"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-cost-per-call",
          "series": "Remaining output (visible-answer estimate)",
          "point": "Claude Haiku 4.5 · Claude Code",
          "metric": "thinking-bill-cost-per-call/Remaining output (visible-answer estimate)",
          "label": "List-price cost per call: reasoning, remaining output and input (calculation) (Remaining output (visible-answer estimate))",
          "value": 0.001795,
          "unit": "usd",
          "display": "$0.0018",
          "n": 24,
          "calculation": true,
          "polarity": "none",
          "context": "Claude Code"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-cost-per-call",
          "series": "Input (prompt, cache priced)",
          "point": "Claude Haiku 4.5 · Claude Code",
          "metric": "thinking-bill-cost-per-call/Input (prompt, cache priced)",
          "label": "List-price cost per call: reasoning, remaining output and input (calculation) (Input (prompt, cache priced))",
          "value": 0.00451,
          "unit": "usd",
          "display": "$0.0045",
          "n": 24,
          "calculation": true,
          "polarity": "none",
          "context": "Claude Code"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-short-vs-hard",
          "series": "Eight hard tasks",
          "point": "Claude Haiku 4.5 · Claude Code",
          "metric": "thinking-bill-short-vs-hard/Eight hard tasks",
          "label": "Reasoning share on short tasks vs hard tasks (calculation) (Eight hard tasks)",
          "value": 91.68,
          "unit": "percent",
          "display": "91.7%",
          "n": 24,
          "range": [
            76.46,
            99.27
          ],
          "spanKind": "minmax",
          "calculation": true,
          "polarity": "none",
          "context": "Claude Code"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-short-vs-hard",
          "series": "Five short tasks",
          "point": "Claude Haiku 4.5 · Claude Code",
          "metric": "thinking-bill-short-vs-hard/Five short tasks",
          "label": "Reasoning share on short tasks vs hard tasks (calculation) (Five short tasks)",
          "value": 90.19,
          "unit": "percent",
          "display": "90.2%",
          "n": 15,
          "range": [
            73.1,
            97.59
          ],
          "spanKind": "minmax",
          "calculation": true,
          "polarity": "none",
          "context": "Claude Code"
        },
        {
          "studySlug": "llm-speed-anatomy",
          "chartId": "speed-anatomy-first-text",
          "series": "Time to first text",
          "point": "Claude Haiku 4.5 · Claude Code",
          "metric": "speed-anatomy-first-text",
          "label": "Time to first text: a 250-line answer, six models",
          "value": 4,
          "unit": "seconds",
          "display": "4.00 s",
          "n": 4,
          "range": [
            2.84,
            6.38
          ],
          "spanKind": "minmax",
          "context": "Claude Code"
        },
        {
          "studySlug": "llm-speed-anatomy",
          "chartId": "speed-anatomy-output-speed",
          "series": "Visible tokens per second",
          "point": "Claude Haiku 4.5 · Claude Code",
          "metric": "speed-anatomy-output-speed",
          "label": "Output speed after the first text: visible tokens per second (calculation)",
          "value": 153.2,
          "unit": "tokens",
          "display": "153",
          "n": 4,
          "range": [
            152.6,
            216.1
          ],
          "spanKind": "minmax",
          "calculation": true,
          "context": "Claude Code"
        },
        {
          "studySlug": "llm-speed-anatomy",
          "chartId": "speed-anatomy-chars-per-second",
          "series": "Characters per second",
          "point": "Claude Haiku 4.5 · Claude Code",
          "metric": "speed-anatomy-chars-per-second",
          "label": "Output speed in characters per second after the first text (calculation)",
          "value": 547,
          "unit": "count",
          "display": "547",
          "n": 3,
          "range": [
            546,
            548
          ],
          "spanKind": "minmax",
          "calculation": true,
          "context": "Claude Code"
        },
        {
          "studySlug": "llm-speed-anatomy",
          "chartId": "speed-anatomy-prompt-size",
          "series": "Claude Haiku 4.5 · Claude Code",
          "point": "1k",
          "metric": "speed-anatomy-prompt-size/1k",
          "label": "Time to first text as the prompt grows: 1k",
          "value": 1.93,
          "unit": "seconds",
          "display": "1.93 s",
          "n": 3,
          "range": [
            1.85,
            2.04
          ],
          "spanKind": "minmax",
          "calculation": true,
          "context": "Claude Code"
        },
        {
          "studySlug": "llm-speed-anatomy",
          "chartId": "speed-anatomy-prompt-size",
          "series": "Claude Haiku 4.5 · Claude Code",
          "point": "16k",
          "metric": "speed-anatomy-prompt-size/16k",
          "label": "Time to first text as the prompt grows: 16k",
          "value": 2.27,
          "unit": "seconds",
          "display": "2.27 s",
          "n": 3,
          "range": [
            2.22,
            2.47
          ],
          "spanKind": "minmax",
          "calculation": true,
          "context": "Claude Code"
        },
        {
          "studySlug": "llm-speed-anatomy",
          "chartId": "speed-anatomy-prompt-size",
          "series": "Claude Haiku 4.5 · Claude Code",
          "point": "64k",
          "metric": "speed-anatomy-prompt-size/64k",
          "label": "Time to first text as the prompt grows: 64k",
          "value": 2.78,
          "unit": "seconds",
          "display": "2.78 s",
          "n": 3,
          "range": [
            2.45,
            2.89
          ],
          "spanKind": "minmax",
          "calculation": true,
          "context": "Claude Code"
        },
        {
          "studySlug": "llm-speed-anatomy",
          "chartId": "speed-anatomy-total-by-size",
          "series": "1k prompt",
          "point": "Claude Haiku 4.5 · Claude Code",
          "metric": "speed-anatomy-total-by-size/1k prompt",
          "label": "Total time per call by prompt size (1k prompt)",
          "value": 2.34,
          "unit": "seconds",
          "display": "2.34 s",
          "n": 3,
          "range": [
            2.22,
            2.46
          ],
          "spanKind": "minmax",
          "context": "Claude Code"
        },
        {
          "studySlug": "llm-speed-anatomy",
          "chartId": "speed-anatomy-total-by-size",
          "series": "16k prompt",
          "point": "Claude Haiku 4.5 · Claude Code",
          "metric": "speed-anatomy-total-by-size/16k prompt",
          "label": "Total time per call by prompt size (16k prompt)",
          "value": 2.79,
          "unit": "seconds",
          "display": "2.79 s",
          "n": 3,
          "range": [
            2.58,
            2.84
          ],
          "spanKind": "minmax",
          "context": "Claude Code"
        },
        {
          "studySlug": "llm-speed-anatomy",
          "chartId": "speed-anatomy-total-by-size",
          "series": "64k prompt",
          "point": "Claude Haiku 4.5 · Claude Code",
          "metric": "speed-anatomy-total-by-size/64k prompt",
          "label": "Total time per call by prompt size (64k prompt)",
          "value": 3.13,
          "unit": "seconds",
          "display": "3.13 s",
          "n": 3,
          "range": [
            2.84,
            3.28
          ],
          "spanKind": "minmax",
          "context": "Claude Code"
        },
        {
          "studySlug": "llm-speed-anatomy",
          "chartId": "speed-anatomy-lookup-correct",
          "series": "Exact answer",
          "point": "Claude Haiku 4.5 · Claude Code",
          "metric": "speed-anatomy-lookup-correct",
          "label": "Exact lookup answers at the 1k, 16k and 64k prompt-size targets",
          "value": 1,
          "unit": "rate",
          "display": "100% (9/9)",
          "n": 9,
          "ci": [
            0.7009,
            1
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Code"
        },
        {
          "studySlug": "haiku-retry-or-escalate",
          "chartId": "retry-escalate-call-cost-by-task",
          "series": "Claude Haiku 4.5 · Claude Code",
          "point": "Interval merge fix",
          "metric": "retry-escalate-call-cost-by-task/Interval merge fix",
          "label": "List-price cost of one call, Haiku 4.5 vs Sonnet 5.5, on each hard task (calculation): Interval merge fix",
          "value": 0.01789,
          "unit": "usd",
          "display": "$0.018",
          "n": 3,
          "calculation": true,
          "context": "Claude Code"
        },
        {
          "studySlug": "haiku-retry-or-escalate",
          "chartId": "retry-escalate-call-cost-by-task",
          "series": "Claude Haiku 4.5 · Claude Code",
          "point": "DST day length",
          "metric": "retry-escalate-call-cost-by-task/DST day length",
          "label": "List-price cost of one call, Haiku 4.5 vs Sonnet 5.5, on each hard task (calculation): DST day length",
          "value": 0.0293,
          "unit": "usd",
          "display": "$0.029",
          "n": 3,
          "calculation": true,
          "context": "Claude Code"
        },
        {
          "studySlug": "haiku-retry-or-escalate",
          "chartId": "retry-escalate-call-cost-by-task",
          "series": "Claude Haiku 4.5 · Claude Code",
          "point": "CSV parser",
          "metric": "retry-escalate-call-cost-by-task/CSV parser",
          "label": "List-price cost of one call, Haiku 4.5 vs Sonnet 5.5, on each hard task (calculation): CSV parser",
          "value": 0.02919,
          "unit": "usd",
          "display": "$0.029",
          "n": 3,
          "calculation": true,
          "context": "Claude Code"
        },
        {
          "studySlug": "haiku-retry-or-escalate",
          "chartId": "retry-escalate-call-cost-by-task",
          "series": "Claude Haiku 4.5 · Claude Code",
          "point": "Event-loop order",
          "metric": "retry-escalate-call-cost-by-task/Event-loop order",
          "label": "List-price cost of one call, Haiku 4.5 vs Sonnet 5.5, on each hard task (calculation): Event-loop order",
          "value": 0.03708,
          "unit": "usd",
          "display": "$0.037",
          "n": 3,
          "calculation": true,
          "context": "Claude Code"
        },
        {
          "studySlug": "haiku-retry-or-escalate",
          "chartId": "retry-escalate-call-cost-by-task",
          "series": "Claude Haiku 4.5 · Claude Code",
          "point": "Room schedule",
          "metric": "retry-escalate-call-cost-by-task/Room schedule",
          "label": "List-price cost of one call, Haiku 4.5 vs Sonnet 5.5, on each hard task (calculation): Room schedule",
          "value": 0.03578,
          "unit": "usd",
          "display": "$0.036",
          "n": 3,
          "calculation": true,
          "context": "Claude Code"
        },
        {
          "studySlug": "haiku-retry-or-escalate",
          "chartId": "retry-escalate-call-cost-by-task",
          "series": "Claude Haiku 4.5 · Claude Code",
          "point": "SemVer regex",
          "metric": "retry-escalate-call-cost-by-task/SemVer regex",
          "label": "List-price cost of one call, Haiku 4.5 vs Sonnet 5.5, on each hard task (calculation): SemVer regex",
          "value": 0.04217,
          "unit": "usd",
          "display": "$0.042",
          "n": 3,
          "calculation": true,
          "context": "Claude Code"
        },
        {
          "studySlug": "haiku-retry-or-escalate",
          "chartId": "retry-escalate-call-cost-by-task",
          "series": "Claude Haiku 4.5 · Claude Code",
          "point": "Money refactor",
          "metric": "retry-escalate-call-cost-by-task/Money refactor",
          "label": "List-price cost of one call, Haiku 4.5 vs Sonnet 5.5, on each hard task (calculation): Money refactor",
          "value": 0.02039,
          "unit": "usd",
          "display": "$0.020",
          "n": 3,
          "calculation": true,
          "context": "Claude Code"
        },
        {
          "studySlug": "haiku-retry-or-escalate",
          "chartId": "retry-escalate-call-cost-by-task",
          "series": "Claude Haiku 4.5 · Claude Code",
          "point": "SQLite report query",
          "metric": "retry-escalate-call-cost-by-task/SQLite report query",
          "label": "List-price cost of one call, Haiku 4.5 vs Sonnet 5.5, on each hard task (calculation): SQLite report query",
          "value": 0.03648,
          "unit": "usd",
          "display": "$0.036",
          "n": 3,
          "calculation": true,
          "context": "Claude Code"
        },
        {
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-pass-rate",
          "series": "Strict pass",
          "point": "Claude Haiku 4.5 · Claude Code",
          "metric": "harder-h2h-pass-rate/Strict pass",
          "label": "Pass rate on 4 harder tasks (Strict pass)",
          "value": 0,
          "unit": "rate",
          "display": "0% (0/12)",
          "n": 12,
          "ci": [
            0,
            0.2425
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Code"
        },
        {
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-pass-rate",
          "series": "Lenient (format misses counted)",
          "point": "Claude Haiku 4.5 · Claude Code",
          "metric": "harder-h2h-pass-rate/Lenient (format misses counted)",
          "label": "Pass rate on 4 harder tasks (Lenient (format misses counted))",
          "value": 0,
          "unit": "rate",
          "display": "0% (0/12)",
          "n": 12,
          "ci": [
            0,
            0.2425
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Code"
        },
        {
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-tool-attempts",
          "series": "Tool attempt",
          "point": "Claude Haiku 4.5 · Claude Code",
          "metric": "harder-h2h-tool-attempts",
          "label": "Calls that tried a tool although tools were off",
          "value": 0.0833,
          "unit": "rate",
          "display": "8% (1/12)",
          "n": 12,
          "ci": [
            0.0149,
            0.3539
          ],
          "spanKind": "ci95",
          "polarity": "none",
          "context": "Claude Code"
        },
        {
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-pass-by-task",
          "series": "Claude Haiku 4.5 · Claude Code",
          "point": "10x10 nonogram",
          "metric": "harder-h2h-pass-by-task/10x10 nonogram",
          "label": "Strict pass rate by task: 10x10 nonogram",
          "value": 0,
          "unit": "rate",
          "display": "0% (0/3)",
          "n": 3,
          "ci": [
            0,
            0.5615
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Code"
        },
        {
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-pass-by-task",
          "series": "Claude Haiku 4.5 · Claude Code",
          "point": "Sudoku, 22 givens",
          "metric": "harder-h2h-pass-by-task/Sudoku, 22 givens",
          "label": "Strict pass rate by task: Sudoku, 22 givens",
          "value": 0,
          "unit": "rate",
          "display": "0% (0/3)",
          "n": 3,
          "ci": [
            0,
            0.5615
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Code"
        },
        {
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-pass-by-task",
          "series": "Claude Haiku 4.5 · Claude Code",
          "point": "6x6 Skyscrapers",
          "metric": "harder-h2h-pass-by-task/6x6 Skyscrapers",
          "label": "Strict pass rate by task: 6x6 Skyscrapers",
          "value": 0,
          "unit": "rate",
          "display": "0% (0/3)",
          "n": 3,
          "ci": [
            0,
            0.5615
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Code"
        },
        {
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-pass-by-task",
          "series": "Claude Haiku 4.5 · Claude Code",
          "point": "Seeded shuffle output",
          "metric": "harder-h2h-pass-by-task/Seeded shuffle output",
          "label": "Strict pass rate by task: Seeded shuffle output",
          "value": 0,
          "unit": "rate",
          "display": "0% (0/3)",
          "n": 3,
          "ci": [
            0,
            0.5615
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Code"
        },
        {
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-total-latency",
          "series": "Total time per call",
          "point": "Claude Haiku 4.5 · Claude Code",
          "metric": "harder-h2h-total-latency",
          "label": "Total time per call on harder tasks",
          "value": 108.98,
          "unit": "seconds",
          "display": "109.0 s",
          "n": 10,
          "range": [
            25.73,
            223.95
          ],
          "spanKind": "minmax",
          "context": "Claude Code"
        },
        {
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-output-tokens",
          "series": "Output tokens",
          "point": "Claude Haiku 4.5 · Claude Code",
          "metric": "harder-h2h-output-tokens/Output tokens",
          "label": "Output tokens per call on harder tasks (Output tokens)",
          "value": 12508,
          "unit": "tokens",
          "display": "12,508",
          "n": 10,
          "range": [
            2965,
            26532
          ],
          "spanKind": "minmax",
          "polarity": "none",
          "context": "Claude Code"
        }
      ]
    },
    {
      "slug": "claude-fable-5-1",
      "name": "Claude Fable 5.1",
      "vendor": "Anthropic",
      "kind": "model",
      "description": "Anthropic’s highest-priced Claude model in these studies, measured through Claude Code.",
      "aliases": [
        "Claude Fable 5.1",
        "Fable 5.1"
      ],
      "facts": [
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-pass-rate",
          "series": "Pass rate",
          "point": "Claude Fable 5.1 · Claude Code",
          "metric": "h2h-pass-rate",
          "label": "Pass rate on five validated tasks",
          "value": 1,
          "unit": "rate",
          "display": "100% (15/15)",
          "n": 15,
          "ci": [
            0.7961,
            1
          ],
          "spanKind": "ci95",
          "context": "Claude Code · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-total-latency",
          "series": "Total time per call",
          "point": "Claude Fable 5.1 · Claude Code",
          "metric": "h2h-total-latency",
          "label": "Total time per call",
          "value": 1.94,
          "unit": "seconds",
          "display": "1.94 s",
          "n": 15,
          "range": [
            1.41,
            9.83
          ],
          "spanKind": "minmax",
          "context": "Claude Code · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-first-useful-latency",
          "series": "Time to first useful output",
          "point": "Claude Fable 5.1 · Claude Code",
          "metric": "h2h-first-useful-latency",
          "label": "Time to first useful output",
          "value": 1.2,
          "unit": "seconds",
          "display": "1.20 s",
          "n": 15,
          "range": [
            0.95,
            7.9
          ],
          "spanKind": "minmax",
          "context": "Claude Code · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-input-tokens",
          "series": "Cache read",
          "point": "Claude Fable 5.1 · Claude Code",
          "metric": "h2h-input-tokens/Cache read",
          "label": "Input tokens per call: what the CLI sends (Cache read)",
          "value": 2760,
          "unit": "tokens",
          "display": "2,760",
          "n": 15,
          "context": "Claude Code · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-input-tokens",
          "series": "Other input",
          "point": "Claude Fable 5.1 · Claude Code",
          "metric": "h2h-input-tokens/Other input",
          "label": "Input tokens per call: what the CLI sends (Other input)",
          "value": 473,
          "unit": "tokens",
          "display": "473",
          "n": 15,
          "context": "Claude Code · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-output-tokens",
          "series": "Output tokens",
          "point": "Claude Fable 5.1 · Claude Code",
          "metric": "h2h-output-tokens/Output tokens",
          "label": "Output tokens per call (Output tokens)",
          "value": 64,
          "unit": "tokens",
          "display": "64",
          "n": 15,
          "context": "Claude Code · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-list-price-per-call",
          "series": "Cost per call",
          "point": "Claude Fable 5.1 · Claude Code",
          "metric": "h2h-list-price-per-call",
          "label": "List-price cost per call (calculation)",
          "value": 0.00987,
          "unit": "usd",
          "display": "$0.0099",
          "n": 15,
          "range": [
            0.0049,
            0.05843
          ],
          "spanKind": "minmax",
          "calculation": true,
          "context": "Claude Code · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-cost-per-pass",
          "series": "Cost per pass",
          "point": "Claude Fable 5.1 · Claude Code",
          "metric": "h2h-cost-per-pass",
          "label": "List-price cost per passing answer (calculation)",
          "value": 0.02054,
          "unit": "usd",
          "display": "$0.021",
          "n": 15,
          "calculation": true,
          "context": "Claude Code · five short validated tasks"
        },
        {
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-pass-rate",
          "series": "Strict pass",
          "point": "Claude Fable 5.1 · Claude Code",
          "metric": "hard-h2h-pass-rate/Strict pass",
          "label": "Pass rate on eight hard tasks (Strict pass)",
          "value": 1,
          "unit": "rate",
          "display": "100% (24/24)",
          "n": 24,
          "ci": [
            0.862,
            1
          ],
          "spanKind": "ci95",
          "context": "Claude Code · eight hard validated tasks"
        },
        {
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-pass-rate",
          "series": "Lenient (format misses counted)",
          "point": "Claude Fable 5.1 · Claude Code",
          "metric": "hard-h2h-pass-rate/Lenient (format misses counted)",
          "label": "Pass rate on eight hard tasks (Lenient (format misses counted))",
          "value": 1,
          "unit": "rate",
          "display": "100% (24/24)",
          "n": 24,
          "ci": [
            0.862,
            1
          ],
          "spanKind": "ci95",
          "context": "Claude Code · eight hard validated tasks"
        },
        {
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-total-latency",
          "series": "Total time per call on hard tasks (separate batches)",
          "point": "Claude Fable 5.1 · Claude Code",
          "metric": "hard-h2h-total-latency",
          "label": "Total time per call on hard tasks (separate batches)",
          "value": 16.13,
          "unit": "seconds",
          "display": "16.1 s",
          "n": 24,
          "range": [
            4.46,
            90
          ],
          "spanKind": "minmax",
          "context": "Claude Code · eight hard validated tasks"
        },
        {
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-first-useful-latency",
          "series": "Time to first useful output on hard tasks",
          "point": "Claude Fable 5.1 · Claude Code",
          "metric": "hard-h2h-first-useful-latency",
          "label": "Time to first useful output on hard tasks",
          "value": 11.63,
          "unit": "seconds",
          "display": "11.6 s",
          "n": 24,
          "range": [
            2,
            85.33
          ],
          "spanKind": "minmax",
          "context": "Claude Code · eight hard validated tasks"
        },
        {
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-output-tokens",
          "series": "Output tokens",
          "point": "Claude Fable 5.1 · Claude Code",
          "metric": "hard-h2h-output-tokens/Output tokens",
          "label": "Output tokens per call on hard tasks (Output tokens)",
          "value": 1366,
          "unit": "tokens",
          "display": "1,366",
          "n": 24,
          "context": "Claude Code · eight hard validated tasks"
        },
        {
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-cost-per-pass",
          "series": "Cost per strict pass",
          "point": "Claude Fable 5.1 · Claude Code",
          "metric": "hard-h2h-cost-per-pass",
          "label": "List-price cost per strict pass on hard tasks (calculation)",
          "value": 0.09331,
          "unit": "usd",
          "display": "$0.093",
          "n": 24,
          "calculation": true,
          "context": "Claude Code · eight hard validated tasks"
        },
        {
          "studySlug": "cost-thought-experiments",
          "chartId": "repriced-cost-per-resolved",
          "series": "Repriced cost per resolved instance",
          "point": "Claude Fable 5.1",
          "metric": "repriced-cost-per-resolved",
          "label": "Thought experiment: the same tokens at other list prices",
          "value": 12.852,
          "unit": "usd",
          "display": "$12.85",
          "calculation": true,
          "context": "calculation: Agent’s recorded tokens at this model’s list price"
        },
        {
          "studySlug": "cost-thought-experiments",
          "chartId": "prompt-cache-savings",
          "series": "With caching (as recorded)",
          "point": "Claude Fable 5.1",
          "metric": "prompt-cache-savings/With caching (as recorded)",
          "label": "Thought experiment: what prompt caching saved (With caching (as recorded))",
          "value": 321.31,
          "unit": "usd",
          "display": "$321.31",
          "calculation": true,
          "context": "calculation: Agent’s recorded tokens at this model’s list price"
        },
        {
          "studySlug": "cost-thought-experiments",
          "chartId": "prompt-cache-savings",
          "series": "Without caching",
          "point": "Claude Fable 5.1",
          "metric": "prompt-cache-savings/Without caching",
          "label": "Thought experiment: what prompt caching saved (Without caching)",
          "value": 1716.65,
          "unit": "usd",
          "display": "$1716.65",
          "calculation": true,
          "context": "calculation: Agent’s recorded tokens at this model’s list price"
        },
        {
          "studySlug": "prompt-cache-break-even",
          "chartId": "cache-break-even-reads",
          "series": "1-hour write (2× input), whole prefix new",
          "point": "Claude Fable 5.1",
          "metric": "cache-break-even-reads/1-hour write (2× input), whole prefix new",
          "label": "Reuses before a cached prefix costs less, by model and write type (calculation) (1-hour write (2× input), whole prefix new)",
          "value": 1.03,
          "unit": "score",
          "display": "1.03",
          "calculation": true,
          "context": "calculation: Agent’s recorded tokens at this model’s list price"
        },
        {
          "studySlug": "prompt-cache-break-even",
          "chartId": "cache-break-even-reads",
          "series": "1-hour write, pooled n = 6 session share, 19% already cached (as recorded)",
          "point": "Claude Fable 5.1",
          "metric": "cache-break-even-reads/1-hour write, pooled n = 6 session share, 19% already cached (as recorded)",
          "label": "Reuses before a cached prefix costs less, by model and write type (calculation) (1-hour write, pooled n = 6 session share, 19% already cached (as recorded))",
          "value": 0.65,
          "unit": "score",
          "display": "0.65",
          "calculation": true,
          "context": "calculation: Agent’s recorded tokens at this model’s list price"
        },
        {
          "studySlug": "prompt-cache-break-even",
          "chartId": "cache-break-even-reads",
          "series": "5-minute write (1.25× input, an assumption)",
          "point": "Claude Fable 5.1",
          "metric": "cache-break-even-reads/5-minute write (1.25× input, an assumption)",
          "label": "Reuses before a cached prefix costs less, by model and write type (calculation) (5-minute write (1.25× input, an assumption))",
          "value": 0.26,
          "unit": "score",
          "display": "0.26",
          "calculation": true,
          "context": "calculation: Agent’s recorded tokens at this model’s list price"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-share",
          "series": "Median call",
          "point": "Claude Fable 5.1 · Claude Code",
          "metric": "thinking-bill-share",
          "label": "Reasoning share of output tokens per call on hard tasks (calculation)",
          "value": 64.24,
          "unit": "percent",
          "display": "64.2%",
          "n": 24,
          "range": [
            23.44,
            97.19
          ],
          "spanKind": "minmax",
          "calculation": true,
          "polarity": "none",
          "context": "Claude Code"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-cost-per-call",
          "series": "Reasoning (output tokens)",
          "point": "Claude Fable 5.1 · Claude Code",
          "metric": "thinking-bill-cost-per-call/Reasoning (output tokens)",
          "label": "List-price cost per call: reasoning, remaining output and input (calculation) (Reasoning (output tokens))",
          "value": 0.053696,
          "unit": "usd",
          "display": "$0.054",
          "n": 24,
          "calculation": true,
          "polarity": "none",
          "context": "Claude Code"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-cost-per-call",
          "series": "Remaining output (visible-answer estimate)",
          "point": "Claude Fable 5.1 · Claude Code",
          "metric": "thinking-bill-cost-per-call/Remaining output (visible-answer estimate)",
          "label": "List-price cost per call: reasoning, remaining output and input (calculation) (Remaining output (visible-answer estimate))",
          "value": 0.018702,
          "unit": "usd",
          "display": "$0.019",
          "n": 24,
          "calculation": true,
          "polarity": "none",
          "context": "Claude Code"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-cost-per-call",
          "series": "Input (prompt, cache priced)",
          "point": "Claude Fable 5.1 · Claude Code",
          "metric": "thinking-bill-cost-per-call/Input (prompt, cache priced)",
          "label": "List-price cost per call: reasoning, remaining output and input (calculation) (Input (prompt, cache priced))",
          "value": 0.02091,
          "unit": "usd",
          "display": "$0.021",
          "n": 24,
          "calculation": true,
          "polarity": "none",
          "context": "Claude Code"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-short-vs-hard",
          "series": "Eight hard tasks",
          "point": "Claude Fable 5.1 · Claude Code",
          "metric": "thinking-bill-short-vs-hard/Eight hard tasks",
          "label": "Reasoning share on short tasks vs hard tasks (calculation) (Eight hard tasks)",
          "value": 64.24,
          "unit": "percent",
          "display": "64.2%",
          "n": 24,
          "range": [
            23.44,
            97.19
          ],
          "spanKind": "minmax",
          "calculation": true,
          "polarity": "none",
          "context": "Claude Code"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-short-vs-hard",
          "series": "Five short tasks",
          "point": "Claude Fable 5.1 · Claude Code",
          "metric": "thinking-bill-short-vs-hard/Five short tasks",
          "label": "Reasoning share on short tasks vs hard tasks (calculation) (Five short tasks)",
          "value": 0,
          "unit": "percent",
          "display": "0%",
          "n": 15,
          "range": [
            0,
            74.01
          ],
          "spanKind": "minmax",
          "calculation": true,
          "polarity": "none",
          "context": "Claude Code"
        },
        {
          "studySlug": "llm-speed-anatomy",
          "chartId": "speed-anatomy-first-text",
          "series": "Time to first text",
          "point": "Claude Fable 5.1 · Claude Code",
          "metric": "speed-anatomy-first-text",
          "label": "Time to first text: a 250-line answer, six models",
          "value": 4.43,
          "unit": "seconds",
          "display": "4.43 s",
          "n": 4,
          "range": [
            2.27,
            4.64
          ],
          "spanKind": "minmax",
          "context": "Claude Code"
        },
        {
          "studySlug": "llm-speed-anatomy",
          "chartId": "speed-anatomy-output-speed",
          "series": "Visible tokens per second",
          "point": "Claude Fable 5.1 · Claude Code",
          "metric": "speed-anatomy-output-speed",
          "label": "Output speed after the first text: visible tokens per second (calculation)",
          "value": 122.6,
          "unit": "tokens",
          "display": "123",
          "n": 4,
          "range": [
            120.9,
            131.4
          ],
          "spanKind": "minmax",
          "calculation": true,
          "context": "Claude Code"
        },
        {
          "studySlug": "llm-speed-anatomy",
          "chartId": "speed-anatomy-chars-per-second",
          "series": "Characters per second",
          "point": "Claude Fable 5.1 · Claude Code",
          "metric": "speed-anatomy-chars-per-second",
          "label": "Output speed in characters per second after the first text (calculation)",
          "value": 273,
          "unit": "count",
          "display": "273",
          "n": 4,
          "range": [
            270,
            293
          ],
          "spanKind": "minmax",
          "calculation": true,
          "context": "Claude Code"
        }
      ]
    },
    {
      "slug": "gpt-6-1-sol-codex-cli",
      "name": "GPT-6.1 Sol (Codex CLI)",
      "vendor": "OpenAI",
      "kind": "model",
      "description": "OpenAI’s GPT-6.1 Sol model run through the Codex CLI at low, medium and high effort.",
      "aliases": [
        "GPT-6.1 Sol · Codex CLI"
      ],
      "facts": [
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-pass-rate",
          "series": "Pass rate",
          "point": "GPT-6.1 Sol (high) · Codex CLI",
          "metric": "h2h-pass-rate",
          "label": "Pass rate on five validated tasks",
          "value": 1,
          "unit": "rate",
          "display": "100% (15/15)",
          "n": 15,
          "ci": [
            0.7961,
            1
          ],
          "spanKind": "ci95",
          "context": "Codex CLI · effort high · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-pass-rate",
          "series": "Pass rate",
          "point": "GPT-6.1 Sol (medium) · Codex CLI",
          "metric": "h2h-pass-rate",
          "label": "Pass rate on five validated tasks",
          "value": 1,
          "unit": "rate",
          "display": "100% (15/15)",
          "n": 15,
          "ci": [
            0.7961,
            1
          ],
          "spanKind": "ci95",
          "context": "Codex CLI · effort medium · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-pass-rate",
          "series": "Pass rate",
          "point": "GPT-6.1 Sol (low) · Codex CLI",
          "metric": "h2h-pass-rate",
          "label": "Pass rate on five validated tasks",
          "value": 1,
          "unit": "rate",
          "display": "100% (10/10)",
          "n": 10,
          "ci": [
            0.7225,
            1
          ],
          "spanKind": "ci95",
          "context": "Codex CLI · effort low · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-total-latency",
          "series": "Total time per call",
          "point": "GPT-6.1 Sol (high) · Codex CLI",
          "metric": "h2h-total-latency",
          "label": "Total time per call",
          "value": 5.6,
          "unit": "seconds",
          "display": "5.60 s",
          "n": 15,
          "range": [
            4.05,
            19.52
          ],
          "spanKind": "minmax",
          "context": "Codex CLI · effort high · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-total-latency",
          "series": "Total time per call",
          "point": "GPT-6.1 Sol (medium) · Codex CLI",
          "metric": "h2h-total-latency",
          "label": "Total time per call",
          "value": 5.65,
          "unit": "seconds",
          "display": "5.65 s",
          "n": 15,
          "range": [
            4.1,
            25.46
          ],
          "spanKind": "minmax",
          "context": "Codex CLI · effort medium · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-total-latency",
          "series": "Total time per call",
          "point": "GPT-6.1 Sol (low) · Codex CLI",
          "metric": "h2h-total-latency",
          "label": "Total time per call",
          "value": 6.26,
          "unit": "seconds",
          "display": "6.26 s",
          "n": 10,
          "range": [
            4.65,
            10.47
          ],
          "spanKind": "minmax",
          "context": "Codex CLI · effort low · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-first-useful-latency",
          "series": "Time to first useful output",
          "point": "GPT-6.1 Sol (high) · Codex CLI",
          "metric": "h2h-first-useful-latency",
          "label": "Time to first useful output",
          "value": 5.32,
          "unit": "seconds",
          "display": "5.32 s",
          "n": 15,
          "range": [
            3.64,
            16.37
          ],
          "spanKind": "minmax",
          "context": "Codex CLI · effort high · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-first-useful-latency",
          "series": "Time to first useful output",
          "point": "GPT-6.1 Sol (medium) · Codex CLI",
          "metric": "h2h-first-useful-latency",
          "label": "Time to first useful output",
          "value": 5.05,
          "unit": "seconds",
          "display": "5.05 s",
          "n": 15,
          "range": [
            3.36,
            17.82
          ],
          "spanKind": "minmax",
          "context": "Codex CLI · effort medium · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-first-useful-latency",
          "series": "Time to first useful output",
          "point": "GPT-6.1 Sol (low) · Codex CLI",
          "metric": "h2h-first-useful-latency",
          "label": "Time to first useful output",
          "value": 5.14,
          "unit": "seconds",
          "display": "5.14 s",
          "n": 10,
          "range": [
            4.02,
            8.5
          ],
          "spanKind": "minmax",
          "context": "Codex CLI · effort low · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-input-tokens",
          "series": "Cache read",
          "point": "GPT-6.1 Sol (high) · Codex CLI",
          "metric": "h2h-input-tokens/Cache read",
          "label": "Input tokens per call: what the CLI sends (Cache read)",
          "value": 6716,
          "unit": "tokens",
          "display": "6,716",
          "n": 15,
          "context": "Codex CLI · effort high · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-input-tokens",
          "series": "Cache read",
          "point": "GPT-6.1 Sol (medium) · Codex CLI",
          "metric": "h2h-input-tokens/Cache read",
          "label": "Input tokens per call: what the CLI sends (Cache read)",
          "value": 5180,
          "unit": "tokens",
          "display": "5,180",
          "n": 15,
          "context": "Codex CLI · effort medium · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-input-tokens",
          "series": "Cache read",
          "point": "GPT-6.1 Sol (low) · Codex CLI",
          "metric": "h2h-input-tokens/Cache read",
          "label": "Input tokens per call: what the CLI sends (Cache read)",
          "value": 8064,
          "unit": "tokens",
          "display": "8,064",
          "n": 10,
          "context": "Codex CLI · effort low · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-input-tokens",
          "series": "Other input",
          "point": "GPT-6.1 Sol (high) · Codex CLI",
          "metric": "h2h-input-tokens/Other input",
          "label": "Input tokens per call: what the CLI sends (Other input)",
          "value": 5406,
          "unit": "tokens",
          "display": "5,406",
          "n": 15,
          "context": "Codex CLI · effort high · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-input-tokens",
          "series": "Other input",
          "point": "GPT-6.1 Sol (medium) · Codex CLI",
          "metric": "h2h-input-tokens/Other input",
          "label": "Input tokens per call: what the CLI sends (Other input)",
          "value": 6943,
          "unit": "tokens",
          "display": "6,943",
          "n": 15,
          "context": "Codex CLI · effort medium · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-input-tokens",
          "series": "Other input",
          "point": "GPT-6.1 Sol (low) · Codex CLI",
          "metric": "h2h-input-tokens/Other input",
          "label": "Input tokens per call: what the CLI sends (Other input)",
          "value": 4059,
          "unit": "tokens",
          "display": "4,059",
          "n": 10,
          "context": "Codex CLI · effort low · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-output-tokens",
          "series": "Output tokens",
          "point": "GPT-6.1 Sol (high) · Codex CLI",
          "metric": "h2h-output-tokens/Output tokens",
          "label": "Output tokens per call (Output tokens)",
          "value": 42,
          "unit": "tokens",
          "display": "42",
          "n": 15,
          "context": "Codex CLI · effort high · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-output-tokens",
          "series": "Output tokens",
          "point": "GPT-6.1 Sol (medium) · Codex CLI",
          "metric": "h2h-output-tokens/Output tokens",
          "label": "Output tokens per call (Output tokens)",
          "value": 42,
          "unit": "tokens",
          "display": "42",
          "n": 15,
          "context": "Codex CLI · effort medium · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-output-tokens",
          "series": "Output tokens",
          "point": "GPT-6.1 Sol (low) · Codex CLI",
          "metric": "h2h-output-tokens/Output tokens",
          "label": "Output tokens per call (Output tokens)",
          "value": 42,
          "unit": "tokens",
          "display": "42",
          "n": 10,
          "context": "Codex CLI · effort low · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-list-price-per-call",
          "series": "Cost per call",
          "point": "GPT-6.1 Sol (high) · Codex CLI",
          "metric": "h2h-list-price-per-call",
          "label": "List-price cost per call (calculation)",
          "value": 0.01047,
          "unit": "usd",
          "display": "$0.010",
          "n": 15,
          "range": [
            0.0066,
            0.02812
          ],
          "spanKind": "minmax",
          "calculation": true,
          "context": "Codex CLI · effort high · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-list-price-per-call",
          "series": "Cost per call",
          "point": "GPT-6.1 Sol (medium) · Codex CLI",
          "metric": "h2h-list-price-per-call",
          "label": "List-price cost per call (calculation)",
          "value": 0.01018,
          "unit": "usd",
          "display": "$0.010",
          "n": 15,
          "range": [
            0.0054,
            0.02686
          ],
          "spanKind": "minmax",
          "calculation": true,
          "context": "Codex CLI · effort medium · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-list-price-per-call",
          "series": "Cost per call",
          "point": "GPT-6.1 Sol (low) · Codex CLI",
          "metric": "h2h-list-price-per-call",
          "label": "List-price cost per call (calculation)",
          "value": 0.00769,
          "unit": "usd",
          "display": "$0.0077",
          "n": 10,
          "range": [
            0.00742,
            0.02649
          ],
          "spanKind": "minmax",
          "calculation": true,
          "context": "Codex CLI · effort low · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-cost-per-pass",
          "series": "Cost per pass",
          "point": "GPT-6.1 Sol (low) · Codex CLI",
          "metric": "h2h-cost-per-pass",
          "label": "List-price cost per passing answer (calculation)",
          "value": 0.00998,
          "unit": "usd",
          "display": "$0.010",
          "n": 10,
          "calculation": true,
          "context": "Codex CLI · effort low · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-cost-per-pass",
          "series": "Cost per pass",
          "point": "GPT-6.1 Sol (high) · Codex CLI",
          "metric": "h2h-cost-per-pass",
          "label": "List-price cost per passing answer (calculation)",
          "value": 0.01322,
          "unit": "usd",
          "display": "$0.013",
          "n": 15,
          "calculation": true,
          "context": "Codex CLI · effort high · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-cost-per-pass",
          "series": "Cost per pass",
          "point": "GPT-6.1 Sol (medium) · Codex CLI",
          "metric": "h2h-cost-per-pass",
          "label": "List-price cost per passing answer (calculation)",
          "value": 0.01564,
          "unit": "usd",
          "display": "$0.016",
          "n": 15,
          "calculation": true,
          "context": "Codex CLI · effort medium · five short validated tasks"
        },
        {
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-pass-rate",
          "series": "Strict pass",
          "point": "GPT-6.1 Sol (medium) · Codex CLI",
          "metric": "hard-h2h-pass-rate/Strict pass",
          "label": "Pass rate on eight hard tasks (Strict pass)",
          "value": 1,
          "unit": "rate",
          "display": "100% (16/16)",
          "n": 16,
          "ci": [
            0.8064,
            1
          ],
          "spanKind": "ci95",
          "context": "Codex CLI · effort medium · eight hard validated tasks"
        },
        {
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-pass-rate",
          "series": "Strict pass",
          "point": "GPT-6.1 Sol (high) · Codex CLI",
          "metric": "hard-h2h-pass-rate/Strict pass",
          "label": "Pass rate on eight hard tasks (Strict pass)",
          "value": 1,
          "unit": "rate",
          "display": "100% (16/16)",
          "n": 16,
          "ci": [
            0.8064,
            1
          ],
          "spanKind": "ci95",
          "context": "Codex CLI · effort high · eight hard validated tasks"
        },
        {
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-pass-rate",
          "series": "Lenient (format misses counted)",
          "point": "GPT-6.1 Sol (medium) · Codex CLI",
          "metric": "hard-h2h-pass-rate/Lenient (format misses counted)",
          "label": "Pass rate on eight hard tasks (Lenient (format misses counted))",
          "value": 1,
          "unit": "rate",
          "display": "100% (16/16)",
          "n": 16,
          "ci": [
            0.8064,
            1
          ],
          "spanKind": "ci95",
          "context": "Codex CLI · effort medium · eight hard validated tasks"
        },
        {
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-pass-rate",
          "series": "Lenient (format misses counted)",
          "point": "GPT-6.1 Sol (high) · Codex CLI",
          "metric": "hard-h2h-pass-rate/Lenient (format misses counted)",
          "label": "Pass rate on eight hard tasks (Lenient (format misses counted))",
          "value": 1,
          "unit": "rate",
          "display": "100% (16/16)",
          "n": 16,
          "ci": [
            0.8064,
            1
          ],
          "spanKind": "ci95",
          "context": "Codex CLI · effort high · eight hard validated tasks"
        },
        {
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-total-latency",
          "series": "Total time per call on hard tasks (separate batches)",
          "point": "GPT-6.1 Sol (medium) · Codex CLI",
          "metric": "hard-h2h-total-latency",
          "label": "Total time per call on hard tasks (separate batches)",
          "value": 13.11,
          "unit": "seconds",
          "display": "13.1 s",
          "n": 16,
          "range": [
            8.54,
            61.6
          ],
          "spanKind": "minmax",
          "context": "Codex CLI · effort medium · eight hard validated tasks"
        },
        {
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-total-latency",
          "series": "Total time per call on hard tasks (separate batches)",
          "point": "GPT-6.1 Sol (high) · Codex CLI",
          "metric": "hard-h2h-total-latency",
          "label": "Total time per call on hard tasks (separate batches)",
          "value": 18.12,
          "unit": "seconds",
          "display": "18.1 s",
          "n": 16,
          "range": [
            11.67,
            92.21
          ],
          "spanKind": "minmax",
          "context": "Codex CLI · effort high · eight hard validated tasks"
        },
        {
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-first-useful-latency",
          "series": "Time to first useful output on hard tasks",
          "point": "GPT-6.1 Sol (medium) · Codex CLI",
          "metric": "hard-h2h-first-useful-latency",
          "label": "Time to first useful output on hard tasks",
          "value": 10.23,
          "unit": "seconds",
          "display": "10.2 s",
          "n": 16,
          "range": [
            6.09,
            40.41
          ],
          "spanKind": "minmax",
          "context": "Codex CLI · effort medium · eight hard validated tasks"
        },
        {
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-first-useful-latency",
          "series": "Time to first useful output on hard tasks",
          "point": "GPT-6.1 Sol (high) · Codex CLI",
          "metric": "hard-h2h-first-useful-latency",
          "label": "Time to first useful output on hard tasks",
          "value": 12.69,
          "unit": "seconds",
          "display": "12.7 s",
          "n": 16,
          "range": [
            8.93,
            75.91
          ],
          "spanKind": "minmax",
          "context": "Codex CLI · effort high · eight hard validated tasks"
        },
        {
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-output-tokens",
          "series": "Output tokens",
          "point": "GPT-6.1 Sol (medium) · Codex CLI",
          "metric": "hard-h2h-output-tokens/Output tokens",
          "label": "Output tokens per call on hard tasks (Output tokens)",
          "value": 335,
          "unit": "tokens",
          "display": "335",
          "n": 16,
          "context": "Codex CLI · effort medium · eight hard validated tasks"
        },
        {
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-output-tokens",
          "series": "Output tokens",
          "point": "GPT-6.1 Sol (high) · Codex CLI",
          "metric": "hard-h2h-output-tokens/Output tokens",
          "label": "Output tokens per call on hard tasks (Output tokens)",
          "value": 436,
          "unit": "tokens",
          "display": "436",
          "n": 16,
          "context": "Codex CLI · effort high · eight hard validated tasks"
        },
        {
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-cost-per-pass",
          "series": "Cost per strict pass",
          "point": "GPT-6.1 Sol (high) · Codex CLI",
          "metric": "hard-h2h-cost-per-pass",
          "label": "List-price cost per strict pass on hard tasks (calculation)",
          "value": 0.01514,
          "unit": "usd",
          "display": "$0.015",
          "n": 16,
          "calculation": true,
          "context": "Codex CLI · effort high · eight hard validated tasks"
        },
        {
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-cost-per-pass",
          "series": "Cost per strict pass",
          "point": "GPT-6.1 Sol (medium) · Codex CLI",
          "metric": "hard-h2h-cost-per-pass",
          "label": "List-price cost per strict pass on hard tasks (calculation)",
          "value": 0.02564,
          "unit": "usd",
          "display": "$0.026",
          "n": 16,
          "calculation": true,
          "context": "Codex CLI · effort medium · eight hard validated tasks"
        },
        {
          "studySlug": "coding-agents-head-to-head",
          "chartId": "coding-agents-pass-rate",
          "series": "Passed every hidden check",
          "point": "GPT-6.1 Sol (medium, tester’s AGENTS.md) · Codex CLI",
          "metric": "coding-agents-pass-rate",
          "label": "Coding sessions that passed every hidden check",
          "value": 1,
          "unit": "rate",
          "display": "100% (12/12)",
          "n": 12,
          "ci": [
            0.7575,
            1
          ],
          "spanKind": "ci95",
          "context": "Codex CLI · effort medium · tester’s AGENTS.md · six small repository tasks with hidden tests"
        },
        {
          "studySlug": "coding-agents-head-to-head",
          "chartId": "coding-agents-wall-time",
          "series": "Wall time per session",
          "point": "GPT-6.1 Sol (medium, tester’s AGENTS.md) · Codex CLI",
          "metric": "coding-agents-wall-time",
          "label": "Time per coding session",
          "value": 113.4,
          "unit": "seconds",
          "display": "113.4 s",
          "n": 12,
          "range": [
            78.5,
            221.9
          ],
          "spanKind": "minmax",
          "context": "Codex CLI · effort medium · tester’s AGENTS.md · six small repository tasks with hidden tests"
        },
        {
          "studySlug": "coding-agents-head-to-head",
          "chartId": "coding-agents-tool-calls",
          "series": "Tool calls per session",
          "point": "GPT-6.1 Sol (medium, tester’s AGENTS.md) · Codex CLI",
          "metric": "coding-agents-tool-calls",
          "label": "Tool calls per coding session",
          "value": 12.5,
          "unit": "calls",
          "display": "12.5",
          "n": 12,
          "range": [
            8,
            18
          ],
          "spanKind": "minmax",
          "context": "Codex CLI · effort medium · tester’s AGENTS.md · six small repository tasks with hidden tests"
        },
        {
          "studySlug": "coding-agents-head-to-head",
          "chartId": "coding-agents-cost-per-pass",
          "series": "List-price cost per pass",
          "point": "GPT-6.1 Sol (medium, tester’s AGENTS.md) · Codex CLI",
          "metric": "coding-agents-cost-per-pass",
          "label": "List-price cost per passing coding session (calculation)",
          "value": 0.0978,
          "unit": "usd",
          "display": "$0.098",
          "n": 12,
          "calculation": true,
          "context": "Codex CLI · effort medium · tester’s AGENTS.md · six small repository tasks with hidden tests"
        },
        {
          "studySlug": "effort-ladder",
          "chartId": "effort-ladder-pass-rate",
          "series": "Strict pass",
          "point": "GPT-6.1 Sol (low) · Codex CLI",
          "metric": "effort-ladder-pass-rate",
          "label": "Strict pass rate by effort on eight hard tasks",
          "value": 1,
          "unit": "rate",
          "display": "100% (16/16)",
          "n": 16,
          "ci": [
            0.8064,
            1
          ],
          "spanKind": "ci95",
          "context": "Codex CLI · effort low · eight hard validated tasks, effort ladder"
        },
        {
          "studySlug": "effort-ladder",
          "chartId": "effort-ladder-pass-rate",
          "series": "Strict pass",
          "point": "GPT-6.1 Sol (medium) · Codex CLI",
          "metric": "effort-ladder-pass-rate",
          "label": "Strict pass rate by effort on eight hard tasks",
          "value": 1,
          "unit": "rate",
          "display": "100% (16/16)",
          "n": 16,
          "ci": [
            0.8064,
            1
          ],
          "spanKind": "ci95",
          "context": "Codex CLI · effort medium · eight hard validated tasks, effort ladder"
        },
        {
          "studySlug": "effort-ladder",
          "chartId": "effort-ladder-pass-rate",
          "series": "Strict pass",
          "point": "GPT-6.1 Sol (high) · Codex CLI",
          "metric": "effort-ladder-pass-rate",
          "label": "Strict pass rate by effort on eight hard tasks",
          "value": 1,
          "unit": "rate",
          "display": "100% (16/16)",
          "n": 16,
          "ci": [
            0.8064,
            1
          ],
          "spanKind": "ci95",
          "context": "Codex CLI · effort high · eight hard validated tasks, effort ladder"
        },
        {
          "studySlug": "effort-ladder",
          "chartId": "effort-ladder-total-latency",
          "series": "Total time per call by effort on hard tasks",
          "point": "GPT-6.1 Sol (low) · Codex CLI",
          "metric": "effort-ladder-total-latency",
          "label": "Total time per call by effort on hard tasks",
          "value": 13.62,
          "unit": "seconds",
          "display": "13.6 s",
          "n": 16,
          "range": [
            7.94,
            44.29
          ],
          "spanKind": "minmax",
          "context": "Codex CLI · effort low · eight hard validated tasks, effort ladder"
        },
        {
          "studySlug": "effort-ladder",
          "chartId": "effort-ladder-total-latency",
          "series": "Total time per call by effort on hard tasks",
          "point": "GPT-6.1 Sol (medium) · Codex CLI",
          "metric": "effort-ladder-total-latency",
          "label": "Total time per call by effort on hard tasks",
          "value": 13.11,
          "unit": "seconds",
          "display": "13.1 s",
          "n": 16,
          "range": [
            8.54,
            61.6
          ],
          "spanKind": "minmax",
          "context": "Codex CLI · effort medium · eight hard validated tasks, effort ladder"
        },
        {
          "studySlug": "effort-ladder",
          "chartId": "effort-ladder-total-latency",
          "series": "Total time per call by effort on hard tasks",
          "point": "GPT-6.1 Sol (high) · Codex CLI",
          "metric": "effort-ladder-total-latency",
          "label": "Total time per call by effort on hard tasks",
          "value": 18.12,
          "unit": "seconds",
          "display": "18.1 s",
          "n": 16,
          "range": [
            11.67,
            92.21
          ],
          "spanKind": "minmax",
          "context": "Codex CLI · effort high · eight hard validated tasks, effort ladder"
        },
        {
          "studySlug": "effort-ladder",
          "chartId": "effort-ladder-output-tokens",
          "series": "Output tokens",
          "point": "GPT-6.1 Sol (low) · Codex CLI",
          "metric": "effort-ladder-output-tokens/Output tokens",
          "label": "Output tokens per call by effort on hard tasks (Output tokens)",
          "value": 284,
          "unit": "tokens",
          "display": "284",
          "n": 16,
          "context": "Codex CLI · effort low · eight hard validated tasks, effort ladder"
        },
        {
          "studySlug": "effort-ladder",
          "chartId": "effort-ladder-output-tokens",
          "series": "Output tokens",
          "point": "GPT-6.1 Sol (medium) · Codex CLI",
          "metric": "effort-ladder-output-tokens/Output tokens",
          "label": "Output tokens per call by effort on hard tasks (Output tokens)",
          "value": 335,
          "unit": "tokens",
          "display": "335",
          "n": 16,
          "context": "Codex CLI · effort medium · eight hard validated tasks, effort ladder"
        },
        {
          "studySlug": "effort-ladder",
          "chartId": "effort-ladder-output-tokens",
          "series": "Output tokens",
          "point": "GPT-6.1 Sol (high) · Codex CLI",
          "metric": "effort-ladder-output-tokens/Output tokens",
          "label": "Output tokens per call by effort on hard tasks (Output tokens)",
          "value": 436,
          "unit": "tokens",
          "display": "436",
          "n": 16,
          "context": "Codex CLI · effort high · eight hard validated tasks, effort ladder"
        },
        {
          "studySlug": "effort-ladder",
          "chartId": "effort-ladder-cost-per-pass",
          "series": "Cost per strict pass",
          "point": "GPT-6.1 Sol (low) · Codex CLI",
          "metric": "effort-ladder-cost-per-pass",
          "label": "List-price cost per strict pass by effort (calculation)",
          "value": 0.01284,
          "unit": "usd",
          "display": "$0.013",
          "n": 16,
          "calculation": true,
          "context": "Codex CLI · effort low · eight hard validated tasks, effort ladder"
        },
        {
          "studySlug": "effort-ladder",
          "chartId": "effort-ladder-cost-per-pass",
          "series": "Cost per strict pass",
          "point": "GPT-6.1 Sol (medium) · Codex CLI",
          "metric": "effort-ladder-cost-per-pass",
          "label": "List-price cost per strict pass by effort (calculation)",
          "value": 0.02564,
          "unit": "usd",
          "display": "$0.026",
          "n": 16,
          "calculation": true,
          "context": "Codex CLI · effort medium · eight hard validated tasks, effort ladder"
        },
        {
          "studySlug": "effort-ladder",
          "chartId": "effort-ladder-cost-per-pass",
          "series": "Cost per strict pass",
          "point": "GPT-6.1 Sol (high) · Codex CLI",
          "metric": "effort-ladder-cost-per-pass",
          "label": "List-price cost per strict pass by effort (calculation)",
          "value": 0.01514,
          "unit": "usd",
          "display": "$0.015",
          "n": 16,
          "calculation": true,
          "context": "Codex CLI · effort high · eight hard validated tasks, effort ladder"
        },
        {
          "studySlug": "caching-consistency",
          "chartId": "consistency-pass-rate",
          "series": "Exact number",
          "point": "GPT-6.1 Sol (medium) · Codex CLI",
          "metric": "consistency-pass-rate/Exact number",
          "label": "Same prompt, 10 times: strict pass rate (Exact number)",
          "value": 1,
          "unit": "rate",
          "display": "100% (10/10)",
          "n": 10,
          "ci": [
            0.7225,
            1
          ],
          "spanKind": "ci95",
          "context": "Codex CLI · effort medium · same prompt repeated 10 times"
        },
        {
          "studySlug": "caching-consistency",
          "chartId": "consistency-pass-rate",
          "series": "JSON object",
          "point": "GPT-6.1 Sol (medium) · Codex CLI",
          "metric": "consistency-pass-rate/JSON object",
          "label": "Same prompt, 10 times: strict pass rate (JSON object)",
          "value": 1,
          "unit": "rate",
          "display": "100% (10/10)",
          "n": 10,
          "ci": [
            0.7225,
            1
          ],
          "spanKind": "ci95",
          "context": "Codex CLI · effort medium · same prompt repeated 10 times"
        },
        {
          "studySlug": "caching-consistency",
          "chartId": "consistency-pass-rate",
          "series": "Code fix",
          "point": "GPT-6.1 Sol (medium) · Codex CLI",
          "metric": "consistency-pass-rate/Code fix",
          "label": "Same prompt, 10 times: strict pass rate (Code fix)",
          "value": 1,
          "unit": "rate",
          "display": "100% (10/10)",
          "n": 10,
          "ci": [
            0.7225,
            1
          ],
          "spanKind": "ci95",
          "context": "Codex CLI · effort medium · same prompt repeated 10 times"
        },
        {
          "studySlug": "caching-consistency",
          "chartId": "consistency-distinct-answers",
          "series": "Exact number",
          "point": "GPT-6.1 Sol (medium) · Codex CLI",
          "metric": "consistency-distinct-answers/Exact number",
          "label": "Same prompt, 10 times: how many different answers (Exact number)",
          "value": 1,
          "unit": "count",
          "display": "1",
          "n": 10,
          "context": "Codex CLI · effort medium · same prompt repeated 10 times"
        },
        {
          "studySlug": "caching-consistency",
          "chartId": "consistency-distinct-answers",
          "series": "JSON object",
          "point": "GPT-6.1 Sol (medium) · Codex CLI",
          "metric": "consistency-distinct-answers/JSON object",
          "label": "Same prompt, 10 times: how many different answers (JSON object)",
          "value": 1,
          "unit": "count",
          "display": "1",
          "n": 10,
          "context": "Codex CLI · effort medium · same prompt repeated 10 times"
        },
        {
          "studySlug": "caching-consistency",
          "chartId": "consistency-distinct-answers",
          "series": "Code fix",
          "point": "GPT-6.1 Sol (medium) · Codex CLI",
          "metric": "consistency-distinct-answers/Code fix",
          "label": "Same prompt, 10 times: how many different answers (Code fix)",
          "value": 6,
          "unit": "count",
          "display": "6",
          "n": 10,
          "context": "Codex CLI · effort medium · same prompt repeated 10 times"
        },
        {
          "studySlug": "caching-consistency",
          "chartId": "consistency-latency-spread",
          "series": "Exact number",
          "point": "GPT-6.1 Sol (medium) · Codex CLI",
          "metric": "consistency-latency-spread/Exact number",
          "label": "Same prompt, 10 times: time per call (Exact number)",
          "value": 13.38,
          "unit": "seconds",
          "display": "13.4 s",
          "n": 10,
          "range": [
            12.29,
            17.97
          ],
          "spanKind": "minmax",
          "context": "Codex CLI · effort medium · same prompt repeated 10 times"
        },
        {
          "studySlug": "caching-consistency",
          "chartId": "consistency-latency-spread",
          "series": "JSON object",
          "point": "GPT-6.1 Sol (medium) · Codex CLI",
          "metric": "consistency-latency-spread/JSON object",
          "label": "Same prompt, 10 times: time per call (JSON object)",
          "value": 6.42,
          "unit": "seconds",
          "display": "6.42 s",
          "n": 10,
          "range": [
            5.25,
            8.26
          ],
          "spanKind": "minmax",
          "context": "Codex CLI · effort medium · same prompt repeated 10 times"
        },
        {
          "studySlug": "caching-consistency",
          "chartId": "consistency-latency-spread",
          "series": "Code fix",
          "point": "GPT-6.1 Sol (medium) · Codex CLI",
          "metric": "consistency-latency-spread/Code fix",
          "label": "Same prompt, 10 times: time per call (Code fix)",
          "value": 11.29,
          "unit": "seconds",
          "display": "11.3 s",
          "n": 10,
          "range": [
            9.08,
            14.85
          ],
          "spanKind": "minmax",
          "context": "Codex CLI · effort medium · same prompt repeated 10 times"
        },
        {
          "studySlug": "cli-model-latency-tokens",
          "chartId": "cli-vs-api-exact-reply-latency",
          "series": "Total time",
          "point": "Codex CLI · GPT-6.1 Sol · low",
          "metric": "cli-vs-api-exact-reply-latency/Total time",
          "label": "CLI vs API: time for a one-line answer (Total time)",
          "value": 4.18,
          "unit": "seconds",
          "display": "4.18 s",
          "n": 5,
          "range": [
            3.86,
            4.53
          ],
          "spanKind": "minmax",
          "context": "Codex CLI · effort low · fixed exact reply, 5 runs"
        },
        {
          "studySlug": "cli-model-latency-tokens",
          "chartId": "cli-vs-api-exact-reply-latency",
          "series": "Total time",
          "point": "Codex CLI · GPT-6.1 Sol · high",
          "metric": "cli-vs-api-exact-reply-latency/Total time",
          "label": "CLI vs API: time for a one-line answer (Total time)",
          "value": 4.19,
          "unit": "seconds",
          "display": "4.19 s",
          "n": 5,
          "range": [
            3.81,
            4.69
          ],
          "spanKind": "minmax",
          "context": "Codex CLI · effort high · fixed exact reply, 5 runs"
        },
        {
          "studySlug": "cli-model-latency-tokens",
          "chartId": "cli-vs-api-exact-reply-latency",
          "series": "First useful output",
          "point": "Codex CLI · GPT-6.1 Sol · low",
          "metric": "cli-vs-api-exact-reply-latency/First useful output",
          "label": "CLI vs API: time for a one-line answer (First useful output)",
          "value": 3.75,
          "unit": "seconds",
          "display": "3.75 s",
          "n": 5,
          "range": [
            3.44,
            4.1
          ],
          "spanKind": "minmax",
          "context": "Codex CLI · effort low · fixed exact reply, 5 runs"
        },
        {
          "studySlug": "cli-model-latency-tokens",
          "chartId": "cli-vs-api-exact-reply-latency",
          "series": "First useful output",
          "point": "Codex CLI · GPT-6.1 Sol · high",
          "metric": "cli-vs-api-exact-reply-latency/First useful output",
          "label": "CLI vs API: time for a one-line answer (First useful output)",
          "value": 3.79,
          "unit": "seconds",
          "display": "3.79 s",
          "n": 5,
          "range": [
            3.37,
            4.3
          ],
          "spanKind": "minmax",
          "context": "Codex CLI · effort high · fixed exact reply, 5 runs"
        },
        {
          "studySlug": "cli-model-latency-tokens",
          "chartId": "cli-vs-api-small-coding-latency",
          "series": "Total time",
          "point": "Codex CLI · GPT-6.1 Sol · low",
          "metric": "cli-vs-api-small-coding-latency/Total time",
          "label": "CLI vs API: time for a small coding task (Total time)",
          "value": 14.15,
          "unit": "seconds",
          "display": "14.2 s",
          "n": 3,
          "range": [
            13.02,
            14.41
          ],
          "spanKind": "minmax",
          "context": "Codex CLI · effort low · small coding task, 3 runs"
        },
        {
          "studySlug": "cli-model-latency-tokens",
          "chartId": "cli-vs-api-small-coding-latency",
          "series": "Total time",
          "point": "Codex CLI · GPT-6.1 Sol · high",
          "metric": "cli-vs-api-small-coding-latency/Total time",
          "label": "CLI vs API: time for a small coding task (Total time)",
          "value": 17.85,
          "unit": "seconds",
          "display": "17.9 s",
          "n": 3,
          "range": [
            17.68,
            22.42
          ],
          "spanKind": "minmax",
          "context": "Codex CLI · effort high · small coding task, 3 runs"
        },
        {
          "studySlug": "cli-model-latency-tokens",
          "chartId": "cli-vs-api-small-coding-latency",
          "series": "First useful output",
          "point": "Codex CLI · GPT-6.1 Sol · low",
          "metric": "cli-vs-api-small-coding-latency/First useful output",
          "label": "CLI vs API: time for a small coding task (First useful output)",
          "value": 13.6,
          "unit": "seconds",
          "display": "13.6 s",
          "n": 3,
          "range": [
            12.52,
            13.83
          ],
          "spanKind": "minmax",
          "context": "Codex CLI · effort low · small coding task, 3 runs"
        },
        {
          "studySlug": "cli-model-latency-tokens",
          "chartId": "cli-vs-api-small-coding-latency",
          "series": "First useful output",
          "point": "Codex CLI · GPT-6.1 Sol · high",
          "metric": "cli-vs-api-small-coding-latency/First useful output",
          "label": "CLI vs API: time for a small coding task (First useful output)",
          "value": 17.27,
          "unit": "seconds",
          "display": "17.3 s",
          "n": 3,
          "range": [
            17.13,
            21.86
          ],
          "spanKind": "minmax",
          "context": "Codex CLI · effort high · small coding task, 3 runs"
        },
        {
          "studySlug": "cli-model-latency-tokens",
          "chartId": "cli-vs-api-prompt-overhead",
          "series": "Input tokens",
          "point": "Codex CLI · GPT-6.1 Sol · low",
          "metric": "cli-vs-api-prompt-overhead",
          "label": "Hidden prompt: input tokens for the same one-line request",
          "value": 19551,
          "unit": "tokens",
          "display": "19,551",
          "n": 5,
          "context": "Codex CLI · effort low · short fixed tasks"
        },
        {
          "studySlug": "cli-model-latency-tokens",
          "chartId": "cli-vs-api-prompt-overhead",
          "series": "Input tokens",
          "point": "Codex CLI · GPT-6.1 Sol · high",
          "metric": "cli-vs-api-prompt-overhead",
          "label": "Hidden prompt: input tokens for the same one-line request",
          "value": 19555,
          "unit": "tokens",
          "display": "19,555",
          "n": 5,
          "context": "Codex CLI · effort high · short fixed tasks"
        },
        {
          "studySlug": "cli-model-latency-tokens",
          "chartId": "scheduler-repair-claude-vs-codex",
          "series": "Total time",
          "point": "Codex CLI · GPT-6.1 Sol · medium",
          "metric": "scheduler-repair-claude-vs-codex/Total time",
          "label": "Repairing a scheduler: Claude Code vs Codex vs API (Total time)",
          "value": 61.16,
          "unit": "seconds",
          "display": "61.2 s",
          "n": 3,
          "range": [
            59.9,
            69.51
          ],
          "spanKind": "minmax",
          "context": "Codex CLI · effort medium · scheduler repair, 296 checks, 3 runs"
        },
        {
          "studySlug": "cli-model-latency-tokens",
          "chartId": "scheduler-repair-claude-vs-codex",
          "series": "First useful output",
          "point": "Codex CLI · GPT-6.1 Sol · medium",
          "metric": "scheduler-repair-claude-vs-codex/First useful output",
          "label": "Repairing a scheduler: Claude Code vs Codex vs API (First useful output)",
          "value": 15.56,
          "unit": "seconds",
          "display": "15.6 s",
          "n": 3,
          "range": [
            13.65,
            23.04
          ],
          "spanKind": "minmax",
          "context": "Codex CLI · effort medium · scheduler repair, 296 checks, 3 runs"
        },
        {
          "studySlug": "cli-model-latency-tokens",
          "chartId": "scheduler-repair-output-tokens",
          "series": "Output tokens",
          "point": "Codex CLI · GPT-6.1 Sol · medium",
          "metric": "scheduler-repair-output-tokens/Output tokens",
          "label": "Output tokens to repair the scheduler (Output tokens)",
          "value": 1181,
          "unit": "tokens",
          "display": "1,181",
          "n": 3,
          "context": "Codex CLI · effort medium · scheduler repair, 296 checks, 3 runs"
        },
        {
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-pass-rate",
          "series": "Strict pass: the whole reply is the right JSON",
          "point": "GPT-6.1 Sol (low, instructions) · Codex CLI",
          "metric": "structured-output-pass-rate/Strict pass: the whole reply is the right JSON",
          "label": "Does a JSON schema raise the pass rate? Instructions vs schema mode (Strict pass: the whole reply is the right JSON)",
          "value": 1,
          "unit": "rate",
          "display": "100% (12/12)",
          "n": 12,
          "ci": [
            0.7575,
            1
          ],
          "spanKind": "ci95",
          "calculation": true,
          "polarity": "higher",
          "context": "Codex CLI · effort low · instructions"
        },
        {
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-pass-rate",
          "series": "Strict pass: the whole reply is the right JSON",
          "point": "GPT-6.1 Sol (low, JSON schema) · Codex CLI",
          "metric": "structured-output-pass-rate/Strict pass: the whole reply is the right JSON",
          "label": "Does a JSON schema raise the pass rate? Instructions vs schema mode (Strict pass: the whole reply is the right JSON)",
          "value": 1,
          "unit": "rate",
          "display": "100% (12/12)",
          "n": 12,
          "ci": [
            0.7575,
            1
          ],
          "spanKind": "ci95",
          "calculation": true,
          "polarity": "higher",
          "context": "Codex CLI · effort low · JSON schema"
        },
        {
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-pass-rate",
          "series": "Right answer in any format (strict pass or format miss)",
          "point": "GPT-6.1 Sol (low, instructions) · Codex CLI",
          "metric": "structured-output-pass-rate/Right answer in any format (strict pass or format miss)",
          "label": "Does a JSON schema raise the pass rate? Instructions vs schema mode (Right answer in any format (strict pass or format miss))",
          "value": 1,
          "unit": "rate",
          "display": "100% (12/12)",
          "n": 12,
          "ci": [
            0.7575,
            1
          ],
          "spanKind": "ci95",
          "calculation": true,
          "polarity": "higher",
          "context": "Codex CLI · effort low · instructions"
        },
        {
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-pass-rate",
          "series": "Right answer in any format (strict pass or format miss)",
          "point": "GPT-6.1 Sol (low, JSON schema) · Codex CLI",
          "metric": "structured-output-pass-rate/Right answer in any format (strict pass or format miss)",
          "label": "Does a JSON schema raise the pass rate? Instructions vs schema mode (Right answer in any format (strict pass or format miss))",
          "value": 1,
          "unit": "rate",
          "display": "100% (12/12)",
          "n": 12,
          "ci": [
            0.7575,
            1
          ],
          "spanKind": "ci95",
          "calculation": true,
          "polarity": "higher",
          "context": "Codex CLI · effort low · JSON schema"
        },
        {
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-outcomes",
          "series": "Strict pass",
          "point": "GPT-6.1 Sol (low, instructions) · Codex CLI",
          "metric": "structured-output-outcomes/Strict pass",
          "label": "What each call produced: strict pass, format miss, wrong values or error (Strict pass)",
          "value": 12,
          "unit": "count",
          "display": "12",
          "n": 12,
          "context": "Codex CLI · effort low · instructions"
        },
        {
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-outcomes",
          "series": "Strict pass",
          "point": "GPT-6.1 Sol (low, JSON schema) · Codex CLI",
          "metric": "structured-output-outcomes/Strict pass",
          "label": "What each call produced: strict pass, format miss, wrong values or error (Strict pass)",
          "value": 12,
          "unit": "count",
          "display": "12",
          "n": 12,
          "context": "Codex CLI · effort low · JSON schema"
        },
        {
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-outcomes",
          "series": "Format miss",
          "point": "GPT-6.1 Sol (low, instructions) · Codex CLI",
          "metric": "structured-output-outcomes/Format miss",
          "label": "What each call produced: strict pass, format miss, wrong values or error (Format miss)",
          "value": 0,
          "unit": "count",
          "display": "0",
          "n": 12,
          "context": "Codex CLI · effort low · instructions"
        },
        {
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-outcomes",
          "series": "Format miss",
          "point": "GPT-6.1 Sol (low, JSON schema) · Codex CLI",
          "metric": "structured-output-outcomes/Format miss",
          "label": "What each call produced: strict pass, format miss, wrong values or error (Format miss)",
          "value": 0,
          "unit": "count",
          "display": "0",
          "n": 12,
          "context": "Codex CLI · effort low · JSON schema"
        },
        {
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-outcomes",
          "series": "Wrong values",
          "point": "GPT-6.1 Sol (low, instructions) · Codex CLI",
          "metric": "structured-output-outcomes/Wrong values",
          "label": "What each call produced: strict pass, format miss, wrong values or error (Wrong values)",
          "value": 0,
          "unit": "count",
          "display": "0",
          "n": 12,
          "context": "Codex CLI · effort low · instructions"
        },
        {
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-outcomes",
          "series": "Wrong values",
          "point": "GPT-6.1 Sol (low, JSON schema) · Codex CLI",
          "metric": "structured-output-outcomes/Wrong values",
          "label": "What each call produced: strict pass, format miss, wrong values or error (Wrong values)",
          "value": 0,
          "unit": "count",
          "display": "0",
          "n": 12,
          "context": "Codex CLI · effort low · JSON schema"
        },
        {
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-outcomes",
          "series": "Error",
          "point": "GPT-6.1 Sol (low, instructions) · Codex CLI",
          "metric": "structured-output-outcomes/Error",
          "label": "What each call produced: strict pass, format miss, wrong values or error (Error)",
          "value": 0,
          "unit": "count",
          "display": "0",
          "n": 12,
          "context": "Codex CLI · effort low · instructions"
        },
        {
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-outcomes",
          "series": "Error",
          "point": "GPT-6.1 Sol (low, JSON schema) · Codex CLI",
          "metric": "structured-output-outcomes/Error",
          "label": "What each call produced: strict pass, format miss, wrong values or error (Error)",
          "value": 0,
          "unit": "count",
          "display": "0",
          "n": 12,
          "context": "Codex CLI · effort low · JSON schema"
        },
        {
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-time",
          "series": "Median time per call (the three prompts pooled)",
          "point": "GPT-6.1 Sol (low, instructions) · Codex CLI",
          "metric": "structured-output-time",
          "label": "Time per call, instructions vs schema mode",
          "value": 6.21,
          "unit": "seconds",
          "display": "6.21 s",
          "n": 12,
          "range": [
            4.2,
            12.27
          ],
          "spanKind": "minmax",
          "context": "Codex CLI · effort low · instructions"
        },
        {
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-time",
          "series": "Median time per call (the three prompts pooled)",
          "point": "GPT-6.1 Sol (low, JSON schema) · Codex CLI",
          "metric": "structured-output-time",
          "label": "Time per call, instructions vs schema mode",
          "value": 5.96,
          "unit": "seconds",
          "display": "5.96 s",
          "n": 12,
          "range": [
            4.62,
            20.97
          ],
          "spanKind": "minmax",
          "context": "Codex CLI · effort low · JSON schema"
        },
        {
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-tokens",
          "series": "Median output tokens per call",
          "point": "GPT-6.1 Sol (low, instructions) · Codex CLI",
          "metric": "structured-output-tokens/Median output tokens per call",
          "label": "Output and reasoning tokens per call, instructions vs schema mode (Median output tokens per call)",
          "value": 117,
          "unit": "tokens",
          "display": "117",
          "n": 12,
          "context": "Codex CLI · effort low · instructions"
        },
        {
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-tokens",
          "series": "Median output tokens per call",
          "point": "GPT-6.1 Sol (low, JSON schema) · Codex CLI",
          "metric": "structured-output-tokens/Median output tokens per call",
          "label": "Output and reasoning tokens per call, instructions vs schema mode (Median output tokens per call)",
          "value": 123,
          "unit": "tokens",
          "display": "123",
          "n": 12,
          "context": "Codex CLI · effort low · JSON schema"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-share",
          "series": "Median call",
          "point": "GPT-6.1 Sol (high) · Codex CLI",
          "metric": "thinking-bill-share",
          "label": "Reasoning share of output tokens per call on hard tasks (calculation)",
          "value": 57.01,
          "unit": "percent",
          "display": "57%",
          "n": 16,
          "range": [
            29.19,
            90.8
          ],
          "spanKind": "minmax",
          "calculation": true,
          "polarity": "none",
          "context": "Codex CLI · effort high"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-share",
          "series": "Median call",
          "point": "GPT-6.1 Sol (medium) · Codex CLI",
          "metric": "thinking-bill-share",
          "label": "Reasoning share of output tokens per call on hard tasks (calculation)",
          "value": 46.33,
          "unit": "percent",
          "display": "46.3%",
          "n": 16,
          "range": [
            11.42,
            86.85
          ],
          "spanKind": "minmax",
          "calculation": true,
          "polarity": "none",
          "context": "Codex CLI · effort medium"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-cost-per-call",
          "series": "Reasoning (output tokens)",
          "point": "GPT-6.1 Sol (medium) · Codex CLI",
          "metric": "thinking-bill-cost-per-call/Reasoning (output tokens)",
          "label": "List-price cost per call: reasoning, remaining output and input (calculation) (Reasoning (output tokens))",
          "value": 0.002273,
          "unit": "usd",
          "display": "$0.0023",
          "n": 16,
          "calculation": true,
          "polarity": "none",
          "context": "Codex CLI · effort medium"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-cost-per-call",
          "series": "Reasoning (output tokens)",
          "point": "GPT-6.1 Sol (high) · Codex CLI",
          "metric": "thinking-bill-cost-per-call/Reasoning (output tokens)",
          "label": "List-price cost per call: reasoning, remaining output and input (calculation) (Reasoning (output tokens))",
          "value": 0.004114,
          "unit": "usd",
          "display": "$0.0041",
          "n": 16,
          "calculation": true,
          "polarity": "none",
          "context": "Codex CLI · effort high"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-cost-per-call",
          "series": "Remaining output (visible-answer estimate)",
          "point": "GPT-6.1 Sol (medium) · Codex CLI",
          "metric": "thinking-bill-cost-per-call/Remaining output (visible-answer estimate)",
          "label": "List-price cost per call: reasoning, remaining output and input (calculation) (Remaining output (visible-answer estimate))",
          "value": 0.003025,
          "unit": "usd",
          "display": "$0.0030",
          "n": 16,
          "calculation": true,
          "polarity": "none",
          "context": "Codex CLI · effort medium"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-cost-per-call",
          "series": "Remaining output (visible-answer estimate)",
          "point": "GPT-6.1 Sol (high) · Codex CLI",
          "metric": "thinking-bill-cost-per-call/Remaining output (visible-answer estimate)",
          "label": "List-price cost per call: reasoning, remaining output and input (calculation) (Remaining output (visible-answer estimate))",
          "value": 0.002905,
          "unit": "usd",
          "display": "$0.0029",
          "n": 16,
          "calculation": true,
          "polarity": "none",
          "context": "Codex CLI · effort high"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-cost-per-call",
          "series": "Input (prompt, cache priced)",
          "point": "GPT-6.1 Sol (medium) · Codex CLI",
          "metric": "thinking-bill-cost-per-call/Input (prompt, cache priced)",
          "label": "List-price cost per call: reasoning, remaining output and input (calculation) (Input (prompt, cache priced))",
          "value": 0.020339,
          "unit": "usd",
          "display": "$0.020",
          "n": 16,
          "calculation": true,
          "polarity": "none",
          "context": "Codex CLI · effort medium"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-cost-per-call",
          "series": "Input (prompt, cache priced)",
          "point": "GPT-6.1 Sol (high) · Codex CLI",
          "metric": "thinking-bill-cost-per-call/Input (prompt, cache priced)",
          "label": "List-price cost per call: reasoning, remaining output and input (calculation) (Input (prompt, cache priced))",
          "value": 0.008117,
          "unit": "usd",
          "display": "$0.0081",
          "n": 16,
          "calculation": true,
          "polarity": "none",
          "context": "Codex CLI · effort high"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-by-effort",
          "series": "Reasoning cost per strict pass",
          "point": "GPT-6.1 Sol (low) · Codex CLI",
          "metric": "thinking-bill-by-effort/Reasoning cost per strict pass",
          "label": "Reasoning cost per strict pass by effort, with the total (calculation) (Reasoning cost per strict pass)",
          "value": 0.001223,
          "unit": "usd",
          "display": "$0.0012",
          "n": 16,
          "calculation": true,
          "polarity": "none",
          "context": "Codex CLI · effort low"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-by-effort",
          "series": "Reasoning cost per strict pass",
          "point": "GPT-6.1 Sol (medium) · Codex CLI",
          "metric": "thinking-bill-by-effort/Reasoning cost per strict pass",
          "label": "Reasoning cost per strict pass by effort, with the total (calculation) (Reasoning cost per strict pass)",
          "value": 0.002273,
          "unit": "usd",
          "display": "$0.0023",
          "n": 16,
          "calculation": true,
          "polarity": "none",
          "context": "Codex CLI · effort medium"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-by-effort",
          "series": "Reasoning cost per strict pass",
          "point": "GPT-6.1 Sol (high) · Codex CLI",
          "metric": "thinking-bill-by-effort/Reasoning cost per strict pass",
          "label": "Reasoning cost per strict pass by effort, with the total (calculation) (Reasoning cost per strict pass)",
          "value": 0.004114,
          "unit": "usd",
          "display": "$0.0041",
          "n": 16,
          "calculation": true,
          "polarity": "none",
          "context": "Codex CLI · effort high"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-by-effort",
          "series": "Total cost per strict pass",
          "point": "GPT-6.1 Sol (low) · Codex CLI",
          "metric": "thinking-bill-by-effort/Total cost per strict pass",
          "label": "Reasoning cost per strict pass by effort, with the total (calculation) (Total cost per strict pass)",
          "value": 0.012837,
          "unit": "usd",
          "display": "$0.013",
          "n": 16,
          "calculation": true,
          "polarity": "none",
          "context": "Codex CLI · effort low"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-by-effort",
          "series": "Total cost per strict pass",
          "point": "GPT-6.1 Sol (medium) · Codex CLI",
          "metric": "thinking-bill-by-effort/Total cost per strict pass",
          "label": "Reasoning cost per strict pass by effort, with the total (calculation) (Total cost per strict pass)",
          "value": 0.025637,
          "unit": "usd",
          "display": "$0.026",
          "n": 16,
          "calculation": true,
          "polarity": "none",
          "context": "Codex CLI · effort medium"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-by-effort",
          "series": "Total cost per strict pass",
          "point": "GPT-6.1 Sol (high) · Codex CLI",
          "metric": "thinking-bill-by-effort/Total cost per strict pass",
          "label": "Reasoning cost per strict pass by effort, with the total (calculation) (Total cost per strict pass)",
          "value": 0.015137,
          "unit": "usd",
          "display": "$0.015",
          "n": 16,
          "calculation": true,
          "polarity": "none",
          "context": "Codex CLI · effort high"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-short-vs-hard",
          "series": "Eight hard tasks",
          "point": "GPT-6.1 Sol (medium) · Codex CLI",
          "metric": "thinking-bill-short-vs-hard/Eight hard tasks",
          "label": "Reasoning share on short tasks vs hard tasks (calculation) (Eight hard tasks)",
          "value": 46.33,
          "unit": "percent",
          "display": "46.3%",
          "n": 16,
          "range": [
            11.42,
            86.85
          ],
          "spanKind": "minmax",
          "calculation": true,
          "polarity": "none",
          "context": "Codex CLI · effort medium"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-short-vs-hard",
          "series": "Eight hard tasks",
          "point": "GPT-6.1 Sol (high) · Codex CLI",
          "metric": "thinking-bill-short-vs-hard/Eight hard tasks",
          "label": "Reasoning share on short tasks vs hard tasks (calculation) (Eight hard tasks)",
          "value": 57.01,
          "unit": "percent",
          "display": "57%",
          "n": 16,
          "range": [
            29.19,
            90.8
          ],
          "spanKind": "minmax",
          "calculation": true,
          "polarity": "none",
          "context": "Codex CLI · effort high"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-short-vs-hard",
          "series": "Five short tasks",
          "point": "GPT-6.1 Sol (medium) · Codex CLI",
          "metric": "thinking-bill-short-vs-hard/Five short tasks",
          "label": "Reasoning share on short tasks vs hard tasks (calculation) (Five short tasks)",
          "value": 41.05,
          "unit": "percent",
          "display": "41%",
          "n": 15,
          "range": [
            0,
            71.43
          ],
          "spanKind": "minmax",
          "calculation": true,
          "polarity": "none",
          "context": "Codex CLI · effort medium"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-short-vs-hard",
          "series": "Five short tasks",
          "point": "GPT-6.1 Sol (high) · Codex CLI",
          "metric": "thinking-bill-short-vs-hard/Five short tasks",
          "label": "Reasoning share on short tasks vs hard tasks (calculation) (Five short tasks)",
          "value": 58.06,
          "unit": "percent",
          "display": "58.1%",
          "n": 15,
          "range": [
            0,
            75.76
          ],
          "spanKind": "minmax",
          "calculation": true,
          "polarity": "none",
          "context": "Codex CLI · effort high"
        },
        {
          "studySlug": "llm-speed-anatomy",
          "chartId": "speed-anatomy-first-text",
          "series": "Time to first text",
          "point": "GPT-6.1 Sol (low) · Codex CLI",
          "metric": "speed-anatomy-first-text",
          "label": "Time to first text: a 250-line answer, six models",
          "value": 3.52,
          "unit": "seconds",
          "display": "3.52 s",
          "n": 4,
          "range": [
            2.75,
            4.42
          ],
          "spanKind": "minmax",
          "context": "Codex CLI · effort low"
        },
        {
          "studySlug": "llm-speed-anatomy",
          "chartId": "speed-anatomy-output-speed",
          "series": "Visible tokens per second",
          "point": "GPT-6.1 Sol (low) · Codex CLI",
          "metric": "speed-anatomy-output-speed",
          "label": "Output speed after the first text: visible tokens per second (calculation)",
          "value": 79.6,
          "unit": "tokens",
          "display": "80",
          "n": 4,
          "range": [
            71.6,
            80.5
          ],
          "spanKind": "minmax",
          "calculation": true,
          "context": "Codex CLI · effort low"
        },
        {
          "studySlug": "llm-speed-anatomy",
          "chartId": "speed-anatomy-chars-per-second",
          "series": "Characters per second",
          "point": "GPT-6.1 Sol (low) · Codex CLI",
          "metric": "speed-anatomy-chars-per-second",
          "label": "Output speed in characters per second after the first text (calculation)",
          "value": 323,
          "unit": "count",
          "display": "323",
          "n": 4,
          "range": [
            291,
            327
          ],
          "spanKind": "minmax",
          "calculation": true,
          "context": "Codex CLI · effort low"
        },
        {
          "studySlug": "llm-speed-anatomy",
          "chartId": "speed-anatomy-prompt-size",
          "series": "GPT-6.1 Sol (low) · Codex CLI",
          "point": "1k",
          "metric": "speed-anatomy-prompt-size/1k",
          "label": "Time to first text as the prompt grows: 1k",
          "value": 3.36,
          "unit": "seconds",
          "display": "3.36 s",
          "n": 3,
          "range": [
            3.36,
            4.75
          ],
          "spanKind": "minmax",
          "calculation": true,
          "context": "Codex CLI · effort low"
        },
        {
          "studySlug": "llm-speed-anatomy",
          "chartId": "speed-anatomy-prompt-size",
          "series": "GPT-6.1 Sol (low) · Codex CLI",
          "point": "16k",
          "metric": "speed-anatomy-prompt-size/16k",
          "label": "Time to first text as the prompt grows: 16k",
          "value": 4.02,
          "unit": "seconds",
          "display": "4.02 s",
          "n": 3,
          "range": [
            3.3,
            4.28
          ],
          "spanKind": "minmax",
          "calculation": true,
          "context": "Codex CLI · effort low"
        },
        {
          "studySlug": "llm-speed-anatomy",
          "chartId": "speed-anatomy-prompt-size",
          "series": "GPT-6.1 Sol (low) · Codex CLI",
          "point": "64k",
          "metric": "speed-anatomy-prompt-size/64k",
          "label": "Time to first text as the prompt grows: 64k",
          "value": 3.93,
          "unit": "seconds",
          "display": "3.93 s",
          "n": 3,
          "range": [
            3.42,
            4.38
          ],
          "spanKind": "minmax",
          "calculation": true,
          "context": "Codex CLI · effort low"
        },
        {
          "studySlug": "llm-speed-anatomy",
          "chartId": "speed-anatomy-total-by-size",
          "series": "1k prompt",
          "point": "GPT-6.1 Sol (low) · Codex CLI",
          "metric": "speed-anatomy-total-by-size/1k prompt",
          "label": "Total time per call by prompt size (1k prompt)",
          "value": 3.43,
          "unit": "seconds",
          "display": "3.43 s",
          "n": 3,
          "range": [
            3.43,
            4.92
          ],
          "spanKind": "minmax",
          "context": "Codex CLI · effort low"
        },
        {
          "studySlug": "llm-speed-anatomy",
          "chartId": "speed-anatomy-total-by-size",
          "series": "16k prompt",
          "point": "GPT-6.1 Sol (low) · Codex CLI",
          "metric": "speed-anatomy-total-by-size/16k prompt",
          "label": "Total time per call by prompt size (16k prompt)",
          "value": 4.14,
          "unit": "seconds",
          "display": "4.14 s",
          "n": 3,
          "range": [
            3.96,
            4.68
          ],
          "spanKind": "minmax",
          "context": "Codex CLI · effort low"
        },
        {
          "studySlug": "llm-speed-anatomy",
          "chartId": "speed-anatomy-total-by-size",
          "series": "64k prompt",
          "point": "GPT-6.1 Sol (low) · Codex CLI",
          "metric": "speed-anatomy-total-by-size/64k prompt",
          "label": "Total time per call by prompt size (64k prompt)",
          "value": 3.96,
          "unit": "seconds",
          "display": "3.96 s",
          "n": 3,
          "range": [
            3.47,
            4.44
          ],
          "spanKind": "minmax",
          "context": "Codex CLI · effort low"
        },
        {
          "studySlug": "llm-speed-anatomy",
          "chartId": "speed-anatomy-lookup-correct",
          "series": "Exact answer",
          "point": "GPT-6.1 Sol (low) · Codex CLI",
          "metric": "speed-anatomy-lookup-correct",
          "label": "Exact lookup answers at the 1k, 16k and 64k prompt-size targets",
          "value": 1,
          "unit": "rate",
          "display": "100% (9/9)",
          "n": 9,
          "ci": [
            0.7009,
            1
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Codex CLI · effort low"
        },
        {
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-pass-rate",
          "series": "Strict pass",
          "point": "GPT-6.1 Sol (medium) · Codex CLI",
          "metric": "harder-h2h-pass-rate/Strict pass",
          "label": "Pass rate on 4 harder tasks (Strict pass)",
          "value": 0.6875,
          "unit": "rate",
          "display": "69% (11/16)",
          "n": 16,
          "ci": [
            0.444,
            0.8584
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Codex CLI · effort medium"
        },
        {
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-pass-rate",
          "series": "Lenient (format misses counted)",
          "point": "GPT-6.1 Sol (medium) · Codex CLI",
          "metric": "harder-h2h-pass-rate/Lenient (format misses counted)",
          "label": "Pass rate on 4 harder tasks (Lenient (format misses counted))",
          "value": 0.6875,
          "unit": "rate",
          "display": "69% (11/16)",
          "n": 16,
          "ci": [
            0.444,
            0.8584
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Codex CLI · effort medium"
        },
        {
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-tool-attempts",
          "series": "Tool attempt",
          "point": "GPT-6.1 Sol (medium) · Codex CLI",
          "metric": "harder-h2h-tool-attempts",
          "label": "Calls that tried a tool although tools were off",
          "value": 0,
          "unit": "rate",
          "display": "0% (0/16)",
          "n": 16,
          "ci": [
            0,
            0.1936
          ],
          "spanKind": "ci95",
          "polarity": "none",
          "context": "Codex CLI · effort medium"
        },
        {
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-pass-by-task",
          "series": "GPT-6.1 Sol (medium) · Codex CLI",
          "point": "10x10 nonogram",
          "metric": "harder-h2h-pass-by-task/10x10 nonogram",
          "label": "Strict pass rate by task: 10x10 nonogram",
          "value": 0.75,
          "unit": "rate",
          "display": "75% (3/4)",
          "n": 4,
          "ci": [
            0.3006,
            0.9544
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Codex CLI · effort medium"
        },
        {
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-pass-by-task",
          "series": "GPT-6.1 Sol (medium) · Codex CLI",
          "point": "Sudoku, 22 givens",
          "metric": "harder-h2h-pass-by-task/Sudoku, 22 givens",
          "label": "Strict pass rate by task: Sudoku, 22 givens",
          "value": 0.25,
          "unit": "rate",
          "display": "25% (1/4)",
          "n": 4,
          "ci": [
            0.0456,
            0.6994
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Codex CLI · effort medium"
        },
        {
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-pass-by-task",
          "series": "GPT-6.1 Sol (medium) · Codex CLI",
          "point": "6x6 Skyscrapers",
          "metric": "harder-h2h-pass-by-task/6x6 Skyscrapers",
          "label": "Strict pass rate by task: 6x6 Skyscrapers",
          "value": 1,
          "unit": "rate",
          "display": "100% (4/4)",
          "n": 4,
          "ci": [
            0.5101,
            1
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Codex CLI · effort medium"
        },
        {
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-pass-by-task",
          "series": "GPT-6.1 Sol (medium) · Codex CLI",
          "point": "Seeded shuffle output",
          "metric": "harder-h2h-pass-by-task/Seeded shuffle output",
          "label": "Strict pass rate by task: Seeded shuffle output",
          "value": 0.75,
          "unit": "rate",
          "display": "75% (3/4)",
          "n": 4,
          "ci": [
            0.3006,
            0.9544
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Codex CLI · effort medium"
        },
        {
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-total-latency",
          "series": "Total time per call",
          "point": "GPT-6.1 Sol (medium) · Codex CLI",
          "metric": "harder-h2h-total-latency",
          "label": "Total time per call on harder tasks",
          "value": 120.24,
          "unit": "seconds",
          "display": "120.2 s",
          "n": 13,
          "range": [
            46.24,
            273.46
          ],
          "spanKind": "minmax",
          "context": "Codex CLI · effort medium"
        },
        {
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-output-tokens",
          "series": "Output tokens",
          "point": "GPT-6.1 Sol (medium) · Codex CLI",
          "metric": "harder-h2h-output-tokens/Output tokens",
          "label": "Output tokens per call on harder tasks (Output tokens)",
          "value": 4994,
          "unit": "tokens",
          "display": "4,994",
          "n": 13,
          "range": [
            2099,
            13413
          ],
          "spanKind": "minmax",
          "polarity": "none",
          "context": "Codex CLI · effort medium"
        },
        {
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-cost-per-pass",
          "series": "Cost per strict pass",
          "point": "GPT-6.1 Sol (medium) · Codex CLI",
          "metric": "harder-h2h-cost-per-pass",
          "label": "List-price cost per strict pass on harder tasks (calculation)",
          "value": 0.08293,
          "unit": "usd",
          "display": "$0.083",
          "n": 16,
          "calculation": true,
          "context": "Codex CLI · effort medium"
        }
      ]
    },
    {
      "slug": "claude-code-cli",
      "name": "Claude Code",
      "vendor": "Anthropic",
      "kind": "cli",
      "description": "Anthropic’s coding CLI. Each measurement pairs it with one Claude model; the context names the model.",
      "aliases": [
        "Claude Code",
        "Claude Code CLI"
      ],
      "facts": [
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-pass-rate",
          "series": "Pass rate",
          "point": "Claude Fable 5.1 · Claude Code",
          "metric": "h2h-pass-rate",
          "label": "Pass rate on five validated tasks",
          "value": 1,
          "unit": "rate",
          "display": "100% (15/15)",
          "n": 15,
          "ci": [
            0.7961,
            1
          ],
          "spanKind": "ci95",
          "context": "Claude Fable 5.1 · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-pass-rate",
          "series": "Pass rate",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "h2h-pass-rate",
          "label": "Pass rate on five validated tasks",
          "value": 0.8,
          "unit": "rate",
          "display": "80% (12/15)",
          "n": 15,
          "ci": [
            0.5481,
            0.9295
          ],
          "spanKind": "ci95",
          "context": "Claude Sonnet 5.5 · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-pass-rate",
          "series": "Pass rate",
          "point": "Claude Opus 5.5 (high) · Claude Code",
          "metric": "h2h-pass-rate",
          "label": "Pass rate on five validated tasks",
          "value": 1,
          "unit": "rate",
          "display": "100% (15/15)",
          "n": 15,
          "ci": [
            0.7961,
            1
          ],
          "spanKind": "ci95",
          "context": "Claude Opus 5.5 · effort high · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-pass-rate",
          "series": "Pass rate",
          "point": "Claude Opus 5.5 · Claude Code",
          "metric": "h2h-pass-rate",
          "label": "Pass rate on five validated tasks",
          "value": 1,
          "unit": "rate",
          "display": "100% (15/15)",
          "n": 15,
          "ci": [
            0.7961,
            1
          ],
          "spanKind": "ci95",
          "context": "Claude Opus 5.5 · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-pass-rate",
          "series": "Pass rate",
          "point": "Claude Opus 5.5 (low) · Claude Code",
          "metric": "h2h-pass-rate",
          "label": "Pass rate on five validated tasks",
          "value": 1,
          "unit": "rate",
          "display": "100% (15/15)",
          "n": 15,
          "ci": [
            0.7961,
            1
          ],
          "spanKind": "ci95",
          "context": "Claude Opus 5.5 · effort low · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-pass-rate",
          "series": "Pass rate",
          "point": "Claude Haiku 4.5 · Claude Code",
          "metric": "h2h-pass-rate",
          "label": "Pass rate on five validated tasks",
          "value": 1,
          "unit": "rate",
          "display": "100% (15/15)",
          "n": 15,
          "ci": [
            0.7961,
            1
          ],
          "spanKind": "ci95",
          "context": "Claude Haiku 4.5 · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-total-latency",
          "series": "Total time per call",
          "point": "Claude Fable 5.1 · Claude Code",
          "metric": "h2h-total-latency",
          "label": "Total time per call",
          "value": 1.94,
          "unit": "seconds",
          "display": "1.94 s",
          "n": 15,
          "range": [
            1.41,
            9.83
          ],
          "spanKind": "minmax",
          "context": "Claude Fable 5.1 · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-total-latency",
          "series": "Total time per call",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "h2h-total-latency",
          "label": "Total time per call",
          "value": 2.31,
          "unit": "seconds",
          "display": "2.31 s",
          "n": 15,
          "range": [
            2.17,
            7.73
          ],
          "spanKind": "minmax",
          "context": "Claude Sonnet 5.5 · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-total-latency",
          "series": "Total time per call",
          "point": "Claude Opus 5.5 (high) · Claude Code",
          "metric": "h2h-total-latency",
          "label": "Total time per call",
          "value": 2.71,
          "unit": "seconds",
          "display": "2.71 s",
          "n": 15,
          "range": [
            2.45,
            11.78
          ],
          "spanKind": "minmax",
          "context": "Claude Opus 5.5 · effort high · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-total-latency",
          "series": "Total time per call",
          "point": "Claude Opus 5.5 · Claude Code",
          "metric": "h2h-total-latency",
          "label": "Total time per call",
          "value": 2.75,
          "unit": "seconds",
          "display": "2.75 s",
          "n": 15,
          "range": [
            2.47,
            8.91
          ],
          "spanKind": "minmax",
          "context": "Claude Opus 5.5 · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-total-latency",
          "series": "Total time per call",
          "point": "Claude Opus 5.5 (low) · Claude Code",
          "metric": "h2h-total-latency",
          "label": "Total time per call",
          "value": 2.83,
          "unit": "seconds",
          "display": "2.83 s",
          "n": 15,
          "range": [
            2.35,
            6.62
          ],
          "spanKind": "minmax",
          "context": "Claude Opus 5.5 · effort low · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-total-latency",
          "series": "Total time per call",
          "point": "Claude Haiku 4.5 · Claude Code",
          "metric": "h2h-total-latency",
          "label": "Total time per call",
          "value": 4.43,
          "unit": "seconds",
          "display": "4.43 s",
          "n": 15,
          "range": [
            3.16,
            23.57
          ],
          "spanKind": "minmax",
          "context": "Claude Haiku 4.5 · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-first-useful-latency",
          "series": "Time to first useful output",
          "point": "Claude Fable 5.1 · Claude Code",
          "metric": "h2h-first-useful-latency",
          "label": "Time to first useful output",
          "value": 1.2,
          "unit": "seconds",
          "display": "1.20 s",
          "n": 15,
          "range": [
            0.95,
            7.9
          ],
          "spanKind": "minmax",
          "context": "Claude Fable 5.1 · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-first-useful-latency",
          "series": "Time to first useful output",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "h2h-first-useful-latency",
          "label": "Time to first useful output",
          "value": 1.56,
          "unit": "seconds",
          "display": "1.56 s",
          "n": 15,
          "range": [
            0.99,
            6.39
          ],
          "spanKind": "minmax",
          "context": "Claude Sonnet 5.5 · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-first-useful-latency",
          "series": "Time to first useful output",
          "point": "Claude Opus 5.5 (high) · Claude Code",
          "metric": "h2h-first-useful-latency",
          "label": "Time to first useful output",
          "value": 2.04,
          "unit": "seconds",
          "display": "2.04 s",
          "n": 15,
          "range": [
            1.4,
            9.94
          ],
          "spanKind": "minmax",
          "context": "Claude Opus 5.5 · effort high · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-first-useful-latency",
          "series": "Time to first useful output",
          "point": "Claude Opus 5.5 · Claude Code",
          "metric": "h2h-first-useful-latency",
          "label": "Time to first useful output",
          "value": 1.92,
          "unit": "seconds",
          "display": "1.92 s",
          "n": 15,
          "range": [
            1.56,
            7.23
          ],
          "spanKind": "minmax",
          "context": "Claude Opus 5.5 · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-first-useful-latency",
          "series": "Time to first useful output",
          "point": "Claude Opus 5.5 (low) · Claude Code",
          "metric": "h2h-first-useful-latency",
          "label": "Time to first useful output",
          "value": 2.39,
          "unit": "seconds",
          "display": "2.39 s",
          "n": 15,
          "range": [
            1.45,
            4.9
          ],
          "spanKind": "minmax",
          "context": "Claude Opus 5.5 · effort low · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-first-useful-latency",
          "series": "Time to first useful output",
          "point": "Claude Haiku 4.5 · Claude Code",
          "metric": "h2h-first-useful-latency",
          "label": "Time to first useful output",
          "value": 3.63,
          "unit": "seconds",
          "display": "3.63 s",
          "n": 15,
          "range": [
            2.78,
            22.27
          ],
          "spanKind": "minmax",
          "context": "Claude Haiku 4.5 · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-input-tokens",
          "series": "Cache read",
          "point": "Claude Fable 5.1 · Claude Code",
          "metric": "h2h-input-tokens/Cache read",
          "label": "Input tokens per call: what the CLI sends (Cache read)",
          "value": 2760,
          "unit": "tokens",
          "display": "2,760",
          "n": 15,
          "context": "Claude Fable 5.1 · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-input-tokens",
          "series": "Cache read",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "h2h-input-tokens/Cache read",
          "label": "Input tokens per call: what the CLI sends (Cache read)",
          "value": 1401,
          "unit": "tokens",
          "display": "1,401",
          "n": 15,
          "context": "Claude Sonnet 5.5 · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-input-tokens",
          "series": "Cache read",
          "point": "Claude Opus 5.5 (high) · Claude Code",
          "metric": "h2h-input-tokens/Cache read",
          "label": "Input tokens per call: what the CLI sends (Cache read)",
          "value": 1463,
          "unit": "tokens",
          "display": "1,463",
          "n": 15,
          "context": "Claude Opus 5.5 · effort high · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-input-tokens",
          "series": "Cache read",
          "point": "Claude Opus 5.5 · Claude Code",
          "metric": "h2h-input-tokens/Cache read",
          "label": "Input tokens per call: what the CLI sends (Cache read)",
          "value": 1401,
          "unit": "tokens",
          "display": "1,401",
          "n": 15,
          "context": "Claude Opus 5.5 · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-input-tokens",
          "series": "Cache read",
          "point": "Claude Opus 5.5 (low) · Claude Code",
          "metric": "h2h-input-tokens/Cache read",
          "label": "Input tokens per call: what the CLI sends (Cache read)",
          "value": 1463,
          "unit": "tokens",
          "display": "1,463",
          "n": 15,
          "context": "Claude Opus 5.5 · effort low · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-input-tokens",
          "series": "Cache read",
          "point": "Claude Haiku 4.5 · Claude Code",
          "metric": "h2h-input-tokens/Cache read",
          "label": "Input tokens per call: what the CLI sends (Cache read)",
          "value": 0,
          "unit": "tokens",
          "display": "0",
          "n": 15,
          "context": "Claude Haiku 4.5 · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-input-tokens",
          "series": "Other input",
          "point": "Claude Fable 5.1 · Claude Code",
          "metric": "h2h-input-tokens/Other input",
          "label": "Input tokens per call: what the CLI sends (Other input)",
          "value": 473,
          "unit": "tokens",
          "display": "473",
          "n": 15,
          "context": "Claude Fable 5.1 · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-input-tokens",
          "series": "Other input",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "h2h-input-tokens/Other input",
          "label": "Input tokens per call: what the CLI sends (Other input)",
          "value": 685,
          "unit": "tokens",
          "display": "685",
          "n": 15,
          "context": "Claude Sonnet 5.5 · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-input-tokens",
          "series": "Other input",
          "point": "Claude Opus 5.5 (high) · Claude Code",
          "metric": "h2h-input-tokens/Other input",
          "label": "Input tokens per call: what the CLI sends (Other input)",
          "value": 619,
          "unit": "tokens",
          "display": "619",
          "n": 15,
          "context": "Claude Opus 5.5 · effort high · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-input-tokens",
          "series": "Other input",
          "point": "Claude Opus 5.5 · Claude Code",
          "metric": "h2h-input-tokens/Other input",
          "label": "Input tokens per call: what the CLI sends (Other input)",
          "value": 680,
          "unit": "tokens",
          "display": "680",
          "n": 15,
          "context": "Claude Opus 5.5 · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-input-tokens",
          "series": "Other input",
          "point": "Claude Opus 5.5 (low) · Claude Code",
          "metric": "h2h-input-tokens/Other input",
          "label": "Input tokens per call: what the CLI sends (Other input)",
          "value": 618,
          "unit": "tokens",
          "display": "618",
          "n": 15,
          "context": "Claude Opus 5.5 · effort low · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-input-tokens",
          "series": "Other input",
          "point": "Claude Haiku 4.5 · Claude Code",
          "metric": "h2h-input-tokens/Other input",
          "label": "Input tokens per call: what the CLI sends (Other input)",
          "value": 3790,
          "unit": "tokens",
          "display": "3,790",
          "n": 15,
          "context": "Claude Haiku 4.5 · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-output-tokens",
          "series": "Output tokens",
          "point": "Claude Fable 5.1 · Claude Code",
          "metric": "h2h-output-tokens/Output tokens",
          "label": "Output tokens per call (Output tokens)",
          "value": 64,
          "unit": "tokens",
          "display": "64",
          "n": 15,
          "context": "Claude Fable 5.1 · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-output-tokens",
          "series": "Output tokens",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "h2h-output-tokens/Output tokens",
          "label": "Output tokens per call (Output tokens)",
          "value": 107,
          "unit": "tokens",
          "display": "107",
          "n": 15,
          "context": "Claude Sonnet 5.5 · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-output-tokens",
          "series": "Output tokens",
          "point": "Claude Opus 5.5 (high) · Claude Code",
          "metric": "h2h-output-tokens/Output tokens",
          "label": "Output tokens per call (Output tokens)",
          "value": 78,
          "unit": "tokens",
          "display": "78",
          "n": 15,
          "context": "Claude Opus 5.5 · effort high · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-output-tokens",
          "series": "Output tokens",
          "point": "Claude Opus 5.5 · Claude Code",
          "metric": "h2h-output-tokens/Output tokens",
          "label": "Output tokens per call (Output tokens)",
          "value": 64,
          "unit": "tokens",
          "display": "64",
          "n": 15,
          "context": "Claude Opus 5.5 · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-output-tokens",
          "series": "Output tokens",
          "point": "Claude Opus 5.5 (low) · Claude Code",
          "metric": "h2h-output-tokens/Output tokens",
          "label": "Output tokens per call (Output tokens)",
          "value": 64,
          "unit": "tokens",
          "display": "64",
          "n": 15,
          "context": "Claude Opus 5.5 · effort low · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-output-tokens",
          "series": "Output tokens",
          "point": "Claude Haiku 4.5 · Claude Code",
          "metric": "h2h-output-tokens/Output tokens",
          "label": "Output tokens per call (Output tokens)",
          "value": 367,
          "unit": "tokens",
          "display": "367",
          "n": 15,
          "context": "Claude Haiku 4.5 · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-list-price-per-call",
          "series": "Cost per call",
          "point": "Claude Fable 5.1 · Claude Code",
          "metric": "h2h-list-price-per-call",
          "label": "List-price cost per call (calculation)",
          "value": 0.00987,
          "unit": "usd",
          "display": "$0.0099",
          "n": 15,
          "range": [
            0.0049,
            0.05843
          ],
          "spanKind": "minmax",
          "calculation": true,
          "context": "Claude Fable 5.1 · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-list-price-per-call",
          "series": "Cost per call",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "h2h-list-price-per-call",
          "label": "List-price cost per call (calculation)",
          "value": 0.0036,
          "unit": "usd",
          "display": "$0.0036",
          "n": 15,
          "range": [
            0.00342,
            0.01021
          ],
          "spanKind": "minmax",
          "calculation": true,
          "context": "Claude Sonnet 5.5 · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-list-price-per-call",
          "series": "Cost per call",
          "point": "Claude Opus 5.5 (high) · Claude Code",
          "metric": "h2h-list-price-per-call",
          "label": "List-price cost per call (calculation)",
          "value": 0.00694,
          "unit": "usd",
          "display": "$0.0069",
          "n": 15,
          "range": [
            0.00592,
            0.02708
          ],
          "spanKind": "minmax",
          "calculation": true,
          "context": "Claude Opus 5.5 · effort high · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-list-price-per-call",
          "series": "Cost per call",
          "point": "Claude Opus 5.5 · Claude Code",
          "metric": "h2h-list-price-per-call",
          "label": "List-price cost per call (calculation)",
          "value": 0.00688,
          "unit": "usd",
          "display": "$0.0069",
          "n": 15,
          "range": [
            0.00592,
            0.02226
          ],
          "spanKind": "minmax",
          "calculation": true,
          "context": "Claude Opus 5.5 · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-list-price-per-call",
          "series": "Cost per call",
          "point": "Claude Opus 5.5 (low) · Claude Code",
          "metric": "h2h-list-price-per-call",
          "label": "List-price cost per call (calculation)",
          "value": 0.00688,
          "unit": "usd",
          "display": "$0.0069",
          "n": 15,
          "range": [
            0.00582,
            0.01793
          ],
          "spanKind": "minmax",
          "calculation": true,
          "context": "Claude Opus 5.5 · effort low · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-list-price-per-call",
          "series": "Cost per call",
          "point": "Claude Haiku 4.5 · Claude Code",
          "metric": "h2h-list-price-per-call",
          "label": "List-price cost per call (calculation)",
          "value": 0.00566,
          "unit": "usd",
          "display": "$0.0057",
          "n": 15,
          "range": [
            0.00513,
            0.01804
          ],
          "spanKind": "minmax",
          "calculation": true,
          "context": "Claude Haiku 4.5 · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-cost-per-pass",
          "series": "Cost per pass",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "h2h-cost-per-pass",
          "label": "List-price cost per passing answer (calculation)",
          "value": 0.00624,
          "unit": "usd",
          "display": "$0.0062",
          "n": 15,
          "calculation": true,
          "context": "Claude Sonnet 5.5 · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-cost-per-pass",
          "series": "Cost per pass",
          "point": "Claude Opus 5.5 (low) · Claude Code",
          "metric": "h2h-cost-per-pass",
          "label": "List-price cost per passing answer (calculation)",
          "value": 0.00829,
          "unit": "usd",
          "display": "$0.0083",
          "n": 15,
          "calculation": true,
          "context": "Claude Opus 5.5 · effort low · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-cost-per-pass",
          "series": "Cost per pass",
          "point": "Claude Haiku 4.5 · Claude Code",
          "metric": "h2h-cost-per-pass",
          "label": "List-price cost per passing answer (calculation)",
          "value": 0.00836,
          "unit": "usd",
          "display": "$0.0084",
          "n": 15,
          "calculation": true,
          "context": "Claude Haiku 4.5 · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-cost-per-pass",
          "series": "Cost per pass",
          "point": "Claude Opus 5.5 · Claude Code",
          "metric": "h2h-cost-per-pass",
          "label": "List-price cost per passing answer (calculation)",
          "value": 0.01009,
          "unit": "usd",
          "display": "$0.010",
          "n": 15,
          "calculation": true,
          "context": "Claude Opus 5.5 · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-cost-per-pass",
          "series": "Cost per pass",
          "point": "Claude Opus 5.5 (high) · Claude Code",
          "metric": "h2h-cost-per-pass",
          "label": "List-price cost per passing answer (calculation)",
          "value": 0.01049,
          "unit": "usd",
          "display": "$0.010",
          "n": 15,
          "calculation": true,
          "context": "Claude Opus 5.5 · effort high · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-cost-per-pass",
          "series": "Cost per pass",
          "point": "Claude Fable 5.1 · Claude Code",
          "metric": "h2h-cost-per-pass",
          "label": "List-price cost per passing answer (calculation)",
          "value": 0.02054,
          "unit": "usd",
          "display": "$0.021",
          "n": 15,
          "calculation": true,
          "context": "Claude Fable 5.1 · five short validated tasks"
        },
        {
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-pass-rate",
          "series": "Strict pass",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "hard-h2h-pass-rate/Strict pass",
          "label": "Pass rate on eight hard tasks (Strict pass)",
          "value": 1,
          "unit": "rate",
          "display": "100% (24/24)",
          "n": 24,
          "ci": [
            0.862,
            1
          ],
          "spanKind": "ci95",
          "context": "Claude Sonnet 5.5 · eight hard validated tasks"
        },
        {
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-pass-rate",
          "series": "Strict pass",
          "point": "Claude Opus 5.5 · Claude Code",
          "metric": "hard-h2h-pass-rate/Strict pass",
          "label": "Pass rate on eight hard tasks (Strict pass)",
          "value": 1,
          "unit": "rate",
          "display": "100% (24/24)",
          "n": 24,
          "ci": [
            0.862,
            1
          ],
          "spanKind": "ci95",
          "context": "Claude Opus 5.5 · eight hard validated tasks"
        },
        {
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-pass-rate",
          "series": "Strict pass",
          "point": "Claude Opus 5.5 (high) · Claude Code",
          "metric": "hard-h2h-pass-rate/Strict pass",
          "label": "Pass rate on eight hard tasks (Strict pass)",
          "value": 1,
          "unit": "rate",
          "display": "100% (24/24)",
          "n": 24,
          "ci": [
            0.862,
            1
          ],
          "spanKind": "ci95",
          "context": "Claude Opus 5.5 · effort high · eight hard validated tasks"
        },
        {
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-pass-rate",
          "series": "Strict pass",
          "point": "Claude Fable 5.1 · Claude Code",
          "metric": "hard-h2h-pass-rate/Strict pass",
          "label": "Pass rate on eight hard tasks (Strict pass)",
          "value": 1,
          "unit": "rate",
          "display": "100% (24/24)",
          "n": 24,
          "ci": [
            0.862,
            1
          ],
          "spanKind": "ci95",
          "context": "Claude Fable 5.1 · eight hard validated tasks"
        },
        {
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-pass-rate",
          "series": "Strict pass",
          "point": "Claude Haiku 4.5 · Claude Code",
          "metric": "hard-h2h-pass-rate/Strict pass",
          "label": "Pass rate on eight hard tasks (Strict pass)",
          "value": 0.4583,
          "unit": "rate",
          "display": "46% (11/24)",
          "n": 24,
          "ci": [
            0.2789,
            0.6493
          ],
          "spanKind": "ci95",
          "context": "Claude Haiku 4.5 · eight hard validated tasks"
        },
        {
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-pass-rate",
          "series": "Lenient (format misses counted)",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "hard-h2h-pass-rate/Lenient (format misses counted)",
          "label": "Pass rate on eight hard tasks (Lenient (format misses counted))",
          "value": 1,
          "unit": "rate",
          "display": "100% (24/24)",
          "n": 24,
          "ci": [
            0.862,
            1
          ],
          "spanKind": "ci95",
          "context": "Claude Sonnet 5.5 · eight hard validated tasks"
        },
        {
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-pass-rate",
          "series": "Lenient (format misses counted)",
          "point": "Claude Opus 5.5 · Claude Code",
          "metric": "hard-h2h-pass-rate/Lenient (format misses counted)",
          "label": "Pass rate on eight hard tasks (Lenient (format misses counted))",
          "value": 1,
          "unit": "rate",
          "display": "100% (24/24)",
          "n": 24,
          "ci": [
            0.862,
            1
          ],
          "spanKind": "ci95",
          "context": "Claude Opus 5.5 · eight hard validated tasks"
        },
        {
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-pass-rate",
          "series": "Lenient (format misses counted)",
          "point": "Claude Opus 5.5 (high) · Claude Code",
          "metric": "hard-h2h-pass-rate/Lenient (format misses counted)",
          "label": "Pass rate on eight hard tasks (Lenient (format misses counted))",
          "value": 1,
          "unit": "rate",
          "display": "100% (24/24)",
          "n": 24,
          "ci": [
            0.862,
            1
          ],
          "spanKind": "ci95",
          "context": "Claude Opus 5.5 · effort high · eight hard validated tasks"
        },
        {
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-pass-rate",
          "series": "Lenient (format misses counted)",
          "point": "Claude Fable 5.1 · Claude Code",
          "metric": "hard-h2h-pass-rate/Lenient (format misses counted)",
          "label": "Pass rate on eight hard tasks (Lenient (format misses counted))",
          "value": 1,
          "unit": "rate",
          "display": "100% (24/24)",
          "n": 24,
          "ci": [
            0.862,
            1
          ],
          "spanKind": "ci95",
          "context": "Claude Fable 5.1 · eight hard validated tasks"
        },
        {
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-pass-rate",
          "series": "Lenient (format misses counted)",
          "point": "Claude Haiku 4.5 · Claude Code",
          "metric": "hard-h2h-pass-rate/Lenient (format misses counted)",
          "label": "Pass rate on eight hard tasks (Lenient (format misses counted))",
          "value": 0.6667,
          "unit": "rate",
          "display": "67% (16/24)",
          "n": 24,
          "ci": [
            0.4671,
            0.8203
          ],
          "spanKind": "ci95",
          "context": "Claude Haiku 4.5 · eight hard validated tasks"
        },
        {
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-total-latency",
          "series": "Total time per call on hard tasks (separate batches)",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "hard-h2h-total-latency",
          "label": "Total time per call on hard tasks (separate batches)",
          "value": 7.75,
          "unit": "seconds",
          "display": "7.75 s",
          "n": 24,
          "range": [
            2.26,
            34.79
          ],
          "spanKind": "minmax",
          "context": "Claude Sonnet 5.5 · eight hard validated tasks"
        },
        {
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-total-latency",
          "series": "Total time per call on hard tasks (separate batches)",
          "point": "Claude Opus 5.5 · Claude Code",
          "metric": "hard-h2h-total-latency",
          "label": "Total time per call on hard tasks (separate batches)",
          "value": 9.18,
          "unit": "seconds",
          "display": "9.18 s",
          "n": 24,
          "range": [
            4.24,
            27.21
          ],
          "spanKind": "minmax",
          "context": "Claude Opus 5.5 · eight hard validated tasks"
        },
        {
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-total-latency",
          "series": "Total time per call on hard tasks (separate batches)",
          "point": "Claude Opus 5.5 (high) · Claude Code",
          "metric": "hard-h2h-total-latency",
          "label": "Total time per call on hard tasks (separate batches)",
          "value": 11.03,
          "unit": "seconds",
          "display": "11.0 s",
          "n": 24,
          "range": [
            3.63,
            63
          ],
          "spanKind": "minmax",
          "context": "Claude Opus 5.5 · effort high · eight hard validated tasks"
        },
        {
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-total-latency",
          "series": "Total time per call on hard tasks (separate batches)",
          "point": "Claude Fable 5.1 · Claude Code",
          "metric": "hard-h2h-total-latency",
          "label": "Total time per call on hard tasks (separate batches)",
          "value": 16.13,
          "unit": "seconds",
          "display": "16.1 s",
          "n": 24,
          "range": [
            4.46,
            90
          ],
          "spanKind": "minmax",
          "context": "Claude Fable 5.1 · eight hard validated tasks"
        },
        {
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-total-latency",
          "series": "Total time per call on hard tasks (separate batches)",
          "point": "Claude Haiku 4.5 · Claude Code",
          "metric": "hard-h2h-total-latency",
          "label": "Total time per call on hard tasks (separate batches)",
          "value": 39.01,
          "unit": "seconds",
          "display": "39.0 s",
          "n": 24,
          "range": [
            15.27,
            75.13
          ],
          "spanKind": "minmax",
          "context": "Claude Haiku 4.5 · eight hard validated tasks"
        },
        {
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-first-useful-latency",
          "series": "Time to first useful output on hard tasks",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "hard-h2h-first-useful-latency",
          "label": "Time to first useful output on hard tasks",
          "value": 5.95,
          "unit": "seconds",
          "display": "5.95 s",
          "n": 24,
          "range": [
            0.86,
            30.57
          ],
          "spanKind": "minmax",
          "context": "Claude Sonnet 5.5 · eight hard validated tasks"
        },
        {
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-first-useful-latency",
          "series": "Time to first useful output on hard tasks",
          "point": "Claude Opus 5.5 · Claude Code",
          "metric": "hard-h2h-first-useful-latency",
          "label": "Time to first useful output on hard tasks",
          "value": 6.78,
          "unit": "seconds",
          "display": "6.78 s",
          "n": 24,
          "range": [
            2.39,
            21.77
          ],
          "spanKind": "minmax",
          "context": "Claude Opus 5.5 · eight hard validated tasks"
        },
        {
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-first-useful-latency",
          "series": "Time to first useful output on hard tasks",
          "point": "Claude Opus 5.5 (high) · Claude Code",
          "metric": "hard-h2h-first-useful-latency",
          "label": "Time to first useful output on hard tasks",
          "value": 7.13,
          "unit": "seconds",
          "display": "7.13 s",
          "n": 24,
          "range": [
            2.15,
            56.23
          ],
          "spanKind": "minmax",
          "context": "Claude Opus 5.5 · effort high · eight hard validated tasks"
        },
        {
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-first-useful-latency",
          "series": "Time to first useful output on hard tasks",
          "point": "Claude Fable 5.1 · Claude Code",
          "metric": "hard-h2h-first-useful-latency",
          "label": "Time to first useful output on hard tasks",
          "value": 11.63,
          "unit": "seconds",
          "display": "11.6 s",
          "n": 24,
          "range": [
            2,
            85.33
          ],
          "spanKind": "minmax",
          "context": "Claude Fable 5.1 · eight hard validated tasks"
        },
        {
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-first-useful-latency",
          "series": "Time to first useful output on hard tasks",
          "point": "Claude Haiku 4.5 · Claude Code",
          "metric": "hard-h2h-first-useful-latency",
          "label": "Time to first useful output on hard tasks",
          "value": 35.54,
          "unit": "seconds",
          "display": "35.5 s",
          "n": 24,
          "range": [
            12.88,
            70.31
          ],
          "spanKind": "minmax",
          "context": "Claude Haiku 4.5 · eight hard validated tasks"
        },
        {
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-output-tokens",
          "series": "Output tokens",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "hard-h2h-output-tokens/Output tokens",
          "label": "Output tokens per call on hard tasks (Output tokens)",
          "value": 1050,
          "unit": "tokens",
          "display": "1,050",
          "n": 24,
          "context": "Claude Sonnet 5.5 · eight hard validated tasks"
        },
        {
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-output-tokens",
          "series": "Output tokens",
          "point": "Claude Opus 5.5 · Claude Code",
          "metric": "hard-h2h-output-tokens/Output tokens",
          "label": "Output tokens per call on hard tasks (Output tokens)",
          "value": 945,
          "unit": "tokens",
          "display": "945",
          "n": 24,
          "context": "Claude Opus 5.5 · eight hard validated tasks"
        },
        {
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-output-tokens",
          "series": "Output tokens",
          "point": "Claude Opus 5.5 (high) · Claude Code",
          "metric": "hard-h2h-output-tokens/Output tokens",
          "label": "Output tokens per call on hard tasks (Output tokens)",
          "value": 1052,
          "unit": "tokens",
          "display": "1,052",
          "n": 24,
          "context": "Claude Opus 5.5 · effort high · eight hard validated tasks"
        },
        {
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-output-tokens",
          "series": "Output tokens",
          "point": "Claude Fable 5.1 · Claude Code",
          "metric": "hard-h2h-output-tokens/Output tokens",
          "label": "Output tokens per call on hard tasks (Output tokens)",
          "value": 1366,
          "unit": "tokens",
          "display": "1,366",
          "n": 24,
          "context": "Claude Fable 5.1 · eight hard validated tasks"
        },
        {
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-output-tokens",
          "series": "Output tokens",
          "point": "Claude Haiku 4.5 · Claude Code",
          "metric": "hard-h2h-output-tokens/Output tokens",
          "label": "Output tokens per call on hard tasks (Output tokens)",
          "value": 5064,
          "unit": "tokens",
          "display": "5,064",
          "n": 24,
          "context": "Claude Haiku 4.5 · eight hard validated tasks"
        },
        {
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-cost-per-pass",
          "series": "Cost per strict pass",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "hard-h2h-cost-per-pass",
          "label": "List-price cost per strict pass on hard tasks (calculation)",
          "value": 0.01435,
          "unit": "usd",
          "display": "$0.014",
          "n": 24,
          "calculation": true,
          "context": "Claude Sonnet 5.5 · eight hard validated tasks"
        },
        {
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-cost-per-pass",
          "series": "Cost per strict pass",
          "point": "Claude Opus 5.5 · Claude Code",
          "metric": "hard-h2h-cost-per-pass",
          "label": "List-price cost per strict pass on hard tasks (calculation)",
          "value": 0.02824,
          "unit": "usd",
          "display": "$0.028",
          "n": 24,
          "calculation": true,
          "context": "Claude Opus 5.5 · eight hard validated tasks"
        },
        {
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-cost-per-pass",
          "series": "Cost per strict pass",
          "point": "Claude Opus 5.5 (high) · Claude Code",
          "metric": "hard-h2h-cost-per-pass",
          "label": "List-price cost per strict pass on hard tasks (calculation)",
          "value": 0.03337,
          "unit": "usd",
          "display": "$0.033",
          "n": 24,
          "calculation": true,
          "context": "Claude Opus 5.5 · effort high · eight hard validated tasks"
        },
        {
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-cost-per-pass",
          "series": "Cost per strict pass",
          "point": "Claude Haiku 4.5 · Claude Code",
          "metric": "hard-h2h-cost-per-pass",
          "label": "List-price cost per strict pass on hard tasks (calculation)",
          "value": 0.0672,
          "unit": "usd",
          "display": "$0.067",
          "n": 24,
          "calculation": true,
          "context": "Claude Haiku 4.5 · eight hard validated tasks"
        },
        {
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-cost-per-pass",
          "series": "Cost per strict pass",
          "point": "Claude Fable 5.1 · Claude Code",
          "metric": "hard-h2h-cost-per-pass",
          "label": "List-price cost per strict pass on hard tasks (calculation)",
          "value": 0.09331,
          "unit": "usd",
          "display": "$0.093",
          "n": 24,
          "calculation": true,
          "context": "Claude Fable 5.1 · eight hard validated tasks"
        },
        {
          "studySlug": "coding-agents-head-to-head",
          "chartId": "coding-agents-pass-rate",
          "series": "Passed every hidden check",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "coding-agents-pass-rate",
          "label": "Coding sessions that passed every hidden check",
          "value": 1,
          "unit": "rate",
          "display": "100% (12/12)",
          "n": 12,
          "ci": [
            0.7575,
            1
          ],
          "spanKind": "ci95",
          "context": "Claude Sonnet 5.5 · six small repository tasks with hidden tests"
        },
        {
          "studySlug": "coding-agents-head-to-head",
          "chartId": "coding-agents-pass-rate",
          "series": "Passed every hidden check",
          "point": "Claude Opus 5.5 · Claude Code",
          "metric": "coding-agents-pass-rate",
          "label": "Coding sessions that passed every hidden check",
          "value": 1,
          "unit": "rate",
          "display": "100% (12/12)",
          "n": 12,
          "ci": [
            0.7575,
            1
          ],
          "spanKind": "ci95",
          "context": "Claude Opus 5.5 · six small repository tasks with hidden tests"
        },
        {
          "studySlug": "coding-agents-head-to-head",
          "chartId": "coding-agents-wall-time",
          "series": "Wall time per session",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "coding-agents-wall-time",
          "label": "Time per coding session",
          "value": 23.1,
          "unit": "seconds",
          "display": "23.1 s",
          "n": 12,
          "range": [
            18.7,
            44.5
          ],
          "spanKind": "minmax",
          "context": "Claude Sonnet 5.5 · six small repository tasks with hidden tests"
        },
        {
          "studySlug": "coding-agents-head-to-head",
          "chartId": "coding-agents-wall-time",
          "series": "Wall time per session",
          "point": "Claude Opus 5.5 · Claude Code",
          "metric": "coding-agents-wall-time",
          "label": "Time per coding session",
          "value": 56.9,
          "unit": "seconds",
          "display": "56.9 s",
          "n": 12,
          "range": [
            29.8,
            185.8
          ],
          "spanKind": "minmax",
          "context": "Claude Opus 5.5 · six small repository tasks with hidden tests"
        },
        {
          "studySlug": "coding-agents-head-to-head",
          "chartId": "coding-agents-tool-calls",
          "series": "Tool calls per session",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "coding-agents-tool-calls",
          "label": "Tool calls per coding session",
          "value": 7.5,
          "unit": "calls",
          "display": "7.5",
          "n": 12,
          "range": [
            3,
            14
          ],
          "spanKind": "minmax",
          "context": "Claude Sonnet 5.5 · six small repository tasks with hidden tests"
        },
        {
          "studySlug": "coding-agents-head-to-head",
          "chartId": "coding-agents-tool-calls",
          "series": "Tool calls per session",
          "point": "Claude Opus 5.5 · Claude Code",
          "metric": "coding-agents-tool-calls",
          "label": "Tool calls per coding session",
          "value": 7.5,
          "unit": "calls",
          "display": "7.5",
          "n": 12,
          "range": [
            5,
            14
          ],
          "spanKind": "minmax",
          "context": "Claude Opus 5.5 · six small repository tasks with hidden tests"
        },
        {
          "studySlug": "coding-agents-head-to-head",
          "chartId": "coding-agents-cost-per-pass",
          "series": "List-price cost per pass",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "coding-agents-cost-per-pass",
          "label": "List-price cost per passing coding session (calculation)",
          "value": 0.085,
          "unit": "usd",
          "display": "$0.085",
          "n": 12,
          "calculation": true,
          "context": "Claude Sonnet 5.5 · six small repository tasks with hidden tests"
        },
        {
          "studySlug": "coding-agents-head-to-head",
          "chartId": "coding-agents-cost-per-pass",
          "series": "List-price cost per pass",
          "point": "Claude Opus 5.5 · Claude Code",
          "metric": "coding-agents-cost-per-pass",
          "label": "List-price cost per passing coding session (calculation)",
          "value": 0.2229,
          "unit": "usd",
          "display": "$0.22",
          "n": 12,
          "calculation": true,
          "context": "Claude Opus 5.5 · six small repository tasks with hidden tests"
        },
        {
          "studySlug": "effort-ladder",
          "chartId": "effort-ladder-pass-rate",
          "series": "Strict pass",
          "point": "Claude Sonnet 5.5 (low) · Claude Code",
          "metric": "effort-ladder-pass-rate",
          "label": "Strict pass rate by effort on eight hard tasks",
          "value": 1,
          "unit": "rate",
          "display": "100% (16/16)",
          "n": 16,
          "ci": [
            0.8064,
            1
          ],
          "spanKind": "ci95",
          "context": "Claude Sonnet 5.5 · effort low · eight hard validated tasks, effort ladder"
        },
        {
          "studySlug": "effort-ladder",
          "chartId": "effort-ladder-pass-rate",
          "series": "Strict pass",
          "point": "Claude Sonnet 5.5 (medium) · Claude Code",
          "metric": "effort-ladder-pass-rate",
          "label": "Strict pass rate by effort on eight hard tasks",
          "value": 1,
          "unit": "rate",
          "display": "100% (16/16)",
          "n": 16,
          "ci": [
            0.8064,
            1
          ],
          "spanKind": "ci95",
          "context": "Claude Sonnet 5.5 · effort medium · eight hard validated tasks, effort ladder"
        },
        {
          "studySlug": "effort-ladder",
          "chartId": "effort-ladder-pass-rate",
          "series": "Strict pass",
          "point": "Claude Sonnet 5.5 (high) · Claude Code",
          "metric": "effort-ladder-pass-rate",
          "label": "Strict pass rate by effort on eight hard tasks",
          "value": 1,
          "unit": "rate",
          "display": "100% (16/16)",
          "n": 16,
          "ci": [
            0.8064,
            1
          ],
          "spanKind": "ci95",
          "context": "Claude Sonnet 5.5 · effort high · eight hard validated tasks, effort ladder"
        },
        {
          "studySlug": "effort-ladder",
          "chartId": "effort-ladder-pass-rate",
          "series": "Strict pass",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "effort-ladder-pass-rate",
          "label": "Strict pass rate by effort on eight hard tasks",
          "value": 1,
          "unit": "rate",
          "display": "100% (16/16)",
          "n": 16,
          "ci": [
            0.8064,
            1
          ],
          "spanKind": "ci95",
          "context": "Claude Sonnet 5.5 · eight hard validated tasks, effort ladder"
        },
        {
          "studySlug": "effort-ladder",
          "chartId": "effort-ladder-pass-rate",
          "series": "Strict pass",
          "point": "Claude Opus 5.5 (low) · Claude Code",
          "metric": "effort-ladder-pass-rate",
          "label": "Strict pass rate by effort on eight hard tasks",
          "value": 1,
          "unit": "rate",
          "display": "100% (16/16)",
          "n": 16,
          "ci": [
            0.8064,
            1
          ],
          "spanKind": "ci95",
          "context": "Claude Opus 5.5 · effort low · eight hard validated tasks, effort ladder"
        },
        {
          "studySlug": "effort-ladder",
          "chartId": "effort-ladder-pass-rate",
          "series": "Strict pass",
          "point": "Claude Opus 5.5 (medium) · Claude Code",
          "metric": "effort-ladder-pass-rate",
          "label": "Strict pass rate by effort on eight hard tasks",
          "value": 1,
          "unit": "rate",
          "display": "100% (16/16)",
          "n": 16,
          "ci": [
            0.8064,
            1
          ],
          "spanKind": "ci95",
          "context": "Claude Opus 5.5 · effort medium · eight hard validated tasks, effort ladder"
        },
        {
          "studySlug": "effort-ladder",
          "chartId": "effort-ladder-pass-rate",
          "series": "Strict pass",
          "point": "Claude Opus 5.5 (high) · Claude Code",
          "metric": "effort-ladder-pass-rate",
          "label": "Strict pass rate by effort on eight hard tasks",
          "value": 1,
          "unit": "rate",
          "display": "100% (16/16)",
          "n": 16,
          "ci": [
            0.8064,
            1
          ],
          "spanKind": "ci95",
          "context": "Claude Opus 5.5 · effort high · eight hard validated tasks, effort ladder"
        },
        {
          "studySlug": "effort-ladder",
          "chartId": "effort-ladder-pass-rate",
          "series": "Strict pass",
          "point": "Claude Opus 5.5 · Claude Code",
          "metric": "effort-ladder-pass-rate",
          "label": "Strict pass rate by effort on eight hard tasks",
          "value": 1,
          "unit": "rate",
          "display": "100% (16/16)",
          "n": 16,
          "ci": [
            0.8064,
            1
          ],
          "spanKind": "ci95",
          "context": "Claude Opus 5.5 · eight hard validated tasks, effort ladder"
        },
        {
          "studySlug": "effort-ladder",
          "chartId": "effort-ladder-total-latency",
          "series": "Total time per call by effort on hard tasks",
          "point": "Claude Sonnet 5.5 (low) · Claude Code",
          "metric": "effort-ladder-total-latency",
          "label": "Total time per call by effort on hard tasks",
          "value": 5.82,
          "unit": "seconds",
          "display": "5.82 s",
          "n": 16,
          "range": [
            2.78,
            19.96
          ],
          "spanKind": "minmax",
          "context": "Claude Sonnet 5.5 · effort low · eight hard validated tasks, effort ladder"
        },
        {
          "studySlug": "effort-ladder",
          "chartId": "effort-ladder-total-latency",
          "series": "Total time per call by effort on hard tasks",
          "point": "Claude Sonnet 5.5 (medium) · Claude Code",
          "metric": "effort-ladder-total-latency",
          "label": "Total time per call by effort on hard tasks",
          "value": 7.63,
          "unit": "seconds",
          "display": "7.63 s",
          "n": 16,
          "range": [
            2.71,
            24.01
          ],
          "spanKind": "minmax",
          "context": "Claude Sonnet 5.5 · effort medium · eight hard validated tasks, effort ladder"
        },
        {
          "studySlug": "effort-ladder",
          "chartId": "effort-ladder-total-latency",
          "series": "Total time per call by effort on hard tasks",
          "point": "Claude Sonnet 5.5 (high) · Claude Code",
          "metric": "effort-ladder-total-latency",
          "label": "Total time per call by effort on hard tasks",
          "value": 8.81,
          "unit": "seconds",
          "display": "8.81 s",
          "n": 16,
          "range": [
            2.93,
            35.81
          ],
          "spanKind": "minmax",
          "context": "Claude Sonnet 5.5 · effort high · eight hard validated tasks, effort ladder"
        },
        {
          "studySlug": "effort-ladder",
          "chartId": "effort-ladder-total-latency",
          "series": "Total time per call by effort on hard tasks",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "effort-ladder-total-latency",
          "label": "Total time per call by effort on hard tasks",
          "value": 7.97,
          "unit": "seconds",
          "display": "7.97 s",
          "n": 16,
          "range": [
            2.26,
            21.61
          ],
          "spanKind": "minmax",
          "context": "Claude Sonnet 5.5 · eight hard validated tasks, effort ladder"
        },
        {
          "studySlug": "effort-ladder",
          "chartId": "effort-ladder-total-latency",
          "series": "Total time per call by effort on hard tasks",
          "point": "Claude Opus 5.5 (low) · Claude Code",
          "metric": "effort-ladder-total-latency",
          "label": "Total time per call by effort on hard tasks",
          "value": 7.5,
          "unit": "seconds",
          "display": "7.50 s",
          "n": 16,
          "range": [
            3.34,
            15.82
          ],
          "spanKind": "minmax",
          "context": "Claude Opus 5.5 · effort low · eight hard validated tasks, effort ladder"
        },
        {
          "studySlug": "effort-ladder",
          "chartId": "effort-ladder-total-latency",
          "series": "Total time per call by effort on hard tasks",
          "point": "Claude Opus 5.5 (medium) · Claude Code",
          "metric": "effort-ladder-total-latency",
          "label": "Total time per call by effort on hard tasks",
          "value": 9.72,
          "unit": "seconds",
          "display": "9.72 s",
          "n": 16,
          "range": [
            4.78,
            31.36
          ],
          "spanKind": "minmax",
          "context": "Claude Opus 5.5 · effort medium · eight hard validated tasks, effort ladder"
        },
        {
          "studySlug": "effort-ladder",
          "chartId": "effort-ladder-total-latency",
          "series": "Total time per call by effort on hard tasks",
          "point": "Claude Opus 5.5 (high) · Claude Code",
          "metric": "effort-ladder-total-latency",
          "label": "Total time per call by effort on hard tasks",
          "value": 10.11,
          "unit": "seconds",
          "display": "10.1 s",
          "n": 16,
          "range": [
            3.63,
            63
          ],
          "spanKind": "minmax",
          "context": "Claude Opus 5.5 · effort high · eight hard validated tasks, effort ladder"
        },
        {
          "studySlug": "effort-ladder",
          "chartId": "effort-ladder-total-latency",
          "series": "Total time per call by effort on hard tasks",
          "point": "Claude Opus 5.5 · Claude Code",
          "metric": "effort-ladder-total-latency",
          "label": "Total time per call by effort on hard tasks",
          "value": 9.18,
          "unit": "seconds",
          "display": "9.18 s",
          "n": 16,
          "range": [
            4.24,
            27.21
          ],
          "spanKind": "minmax",
          "context": "Claude Opus 5.5 · eight hard validated tasks, effort ladder"
        },
        {
          "studySlug": "effort-ladder",
          "chartId": "effort-ladder-output-tokens",
          "series": "Output tokens",
          "point": "Claude Sonnet 5.5 (low) · Claude Code",
          "metric": "effort-ladder-output-tokens/Output tokens",
          "label": "Output tokens per call by effort on hard tasks (Output tokens)",
          "value": 667,
          "unit": "tokens",
          "display": "667",
          "n": 16,
          "context": "Claude Sonnet 5.5 · effort low · eight hard validated tasks, effort ladder"
        },
        {
          "studySlug": "effort-ladder",
          "chartId": "effort-ladder-output-tokens",
          "series": "Output tokens",
          "point": "Claude Sonnet 5.5 (medium) · Claude Code",
          "metric": "effort-ladder-output-tokens/Output tokens",
          "label": "Output tokens per call by effort on hard tasks (Output tokens)",
          "value": 770,
          "unit": "tokens",
          "display": "770",
          "n": 16,
          "context": "Claude Sonnet 5.5 · effort medium · eight hard validated tasks, effort ladder"
        },
        {
          "studySlug": "effort-ladder",
          "chartId": "effort-ladder-output-tokens",
          "series": "Output tokens",
          "point": "Claude Sonnet 5.5 (high) · Claude Code",
          "metric": "effort-ladder-output-tokens/Output tokens",
          "label": "Output tokens per call by effort on hard tasks (Output tokens)",
          "value": 1192,
          "unit": "tokens",
          "display": "1,192",
          "n": 16,
          "context": "Claude Sonnet 5.5 · effort high · eight hard validated tasks, effort ladder"
        },
        {
          "studySlug": "effort-ladder",
          "chartId": "effort-ladder-output-tokens",
          "series": "Output tokens",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "effort-ladder-output-tokens/Output tokens",
          "label": "Output tokens per call by effort on hard tasks (Output tokens)",
          "value": 1054,
          "unit": "tokens",
          "display": "1,054",
          "n": 16,
          "context": "Claude Sonnet 5.5 · eight hard validated tasks, effort ladder"
        },
        {
          "studySlug": "effort-ladder",
          "chartId": "effort-ladder-output-tokens",
          "series": "Output tokens",
          "point": "Claude Opus 5.5 (low) · Claude Code",
          "metric": "effort-ladder-output-tokens/Output tokens",
          "label": "Output tokens per call by effort on hard tasks (Output tokens)",
          "value": 594,
          "unit": "tokens",
          "display": "594",
          "n": 16,
          "context": "Claude Opus 5.5 · effort low · eight hard validated tasks, effort ladder"
        },
        {
          "studySlug": "effort-ladder",
          "chartId": "effort-ladder-output-tokens",
          "series": "Output tokens",
          "point": "Claude Opus 5.5 (medium) · Claude Code",
          "metric": "effort-ladder-output-tokens/Output tokens",
          "label": "Output tokens per call by effort on hard tasks (Output tokens)",
          "value": 853,
          "unit": "tokens",
          "display": "853",
          "n": 16,
          "context": "Claude Opus 5.5 · effort medium · eight hard validated tasks, effort ladder"
        },
        {
          "studySlug": "effort-ladder",
          "chartId": "effort-ladder-output-tokens",
          "series": "Output tokens",
          "point": "Claude Opus 5.5 (high) · Claude Code",
          "metric": "effort-ladder-output-tokens/Output tokens",
          "label": "Output tokens per call by effort on hard tasks (Output tokens)",
          "value": 1052,
          "unit": "tokens",
          "display": "1,052",
          "n": 16,
          "context": "Claude Opus 5.5 · effort high · eight hard validated tasks, effort ladder"
        },
        {
          "studySlug": "effort-ladder",
          "chartId": "effort-ladder-output-tokens",
          "series": "Output tokens",
          "point": "Claude Opus 5.5 · Claude Code",
          "metric": "effort-ladder-output-tokens/Output tokens",
          "label": "Output tokens per call by effort on hard tasks (Output tokens)",
          "value": 945,
          "unit": "tokens",
          "display": "945",
          "n": 16,
          "context": "Claude Opus 5.5 · eight hard validated tasks, effort ladder"
        },
        {
          "studySlug": "effort-ladder",
          "chartId": "effort-ladder-cost-per-pass",
          "series": "Cost per strict pass",
          "point": "Claude Sonnet 5.5 (low) · Claude Code",
          "metric": "effort-ladder-cost-per-pass",
          "label": "List-price cost per strict pass by effort (calculation)",
          "value": 0.01219,
          "unit": "usd",
          "display": "$0.012",
          "n": 16,
          "calculation": true,
          "context": "Claude Sonnet 5.5 · effort low · eight hard validated tasks, effort ladder"
        },
        {
          "studySlug": "effort-ladder",
          "chartId": "effort-ladder-cost-per-pass",
          "series": "Cost per strict pass",
          "point": "Claude Sonnet 5.5 (medium) · Claude Code",
          "metric": "effort-ladder-cost-per-pass",
          "label": "List-price cost per strict pass by effort (calculation)",
          "value": 0.01352,
          "unit": "usd",
          "display": "$0.014",
          "n": 16,
          "calculation": true,
          "context": "Claude Sonnet 5.5 · effort medium · eight hard validated tasks, effort ladder"
        },
        {
          "studySlug": "effort-ladder",
          "chartId": "effort-ladder-cost-per-pass",
          "series": "Cost per strict pass",
          "point": "Claude Sonnet 5.5 (high) · Claude Code",
          "metric": "effort-ladder-cost-per-pass",
          "label": "List-price cost per strict pass by effort (calculation)",
          "value": 0.01671,
          "unit": "usd",
          "display": "$0.017",
          "n": 16,
          "calculation": true,
          "context": "Claude Sonnet 5.5 · effort high · eight hard validated tasks, effort ladder"
        },
        {
          "studySlug": "effort-ladder",
          "chartId": "effort-ladder-cost-per-pass",
          "series": "Cost per strict pass",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "effort-ladder-cost-per-pass",
          "label": "List-price cost per strict pass by effort (calculation)",
          "value": 0.01398,
          "unit": "usd",
          "display": "$0.014",
          "n": 16,
          "calculation": true,
          "context": "Claude Sonnet 5.5 · eight hard validated tasks, effort ladder"
        },
        {
          "studySlug": "effort-ladder",
          "chartId": "effort-ladder-cost-per-pass",
          "series": "Cost per strict pass",
          "point": "Claude Opus 5.5 (low) · Claude Code",
          "metric": "effort-ladder-cost-per-pass",
          "label": "List-price cost per strict pass by effort (calculation)",
          "value": 0.02115,
          "unit": "usd",
          "display": "$0.021",
          "n": 16,
          "calculation": true,
          "context": "Claude Opus 5.5 · effort low · eight hard validated tasks, effort ladder"
        },
        {
          "studySlug": "effort-ladder",
          "chartId": "effort-ladder-cost-per-pass",
          "series": "Cost per strict pass",
          "point": "Claude Opus 5.5 (medium) · Claude Code",
          "metric": "effort-ladder-cost-per-pass",
          "label": "List-price cost per strict pass by effort (calculation)",
          "value": 0.02947,
          "unit": "usd",
          "display": "$0.029",
          "n": 16,
          "calculation": true,
          "context": "Claude Opus 5.5 · effort medium · eight hard validated tasks, effort ladder"
        },
        {
          "studySlug": "effort-ladder",
          "chartId": "effort-ladder-cost-per-pass",
          "series": "Cost per strict pass",
          "point": "Claude Opus 5.5 (high) · Claude Code",
          "metric": "effort-ladder-cost-per-pass",
          "label": "List-price cost per strict pass by effort (calculation)",
          "value": 0.03368,
          "unit": "usd",
          "display": "$0.034",
          "n": 16,
          "calculation": true,
          "context": "Claude Opus 5.5 · effort high · eight hard validated tasks, effort ladder"
        },
        {
          "studySlug": "effort-ladder",
          "chartId": "effort-ladder-cost-per-pass",
          "series": "Cost per strict pass",
          "point": "Claude Opus 5.5 · Claude Code",
          "metric": "effort-ladder-cost-per-pass",
          "label": "List-price cost per strict pass by effort (calculation)",
          "value": 0.02893,
          "unit": "usd",
          "display": "$0.029",
          "n": 16,
          "calculation": true,
          "context": "Claude Opus 5.5 · eight hard validated tasks, effort ladder"
        },
        {
          "studySlug": "caching-consistency",
          "chartId": "caching-cost-with-without",
          "series": "With the cache, as recorded",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "caching-cost-with-without/With the cache, as recorded",
          "label": "List-price cost of 5-question sessions with and without the cache (calculation) (With the cache, as recorded)",
          "value": 0.135003,
          "unit": "usd",
          "display": "$0.14",
          "n": 15,
          "calculation": true,
          "context": "Claude Sonnet 5.5 · calculation: 5-turn cached sessions over a fixed ledger"
        },
        {
          "studySlug": "caching-consistency",
          "chartId": "caching-cost-with-without",
          "series": "With the cache, as recorded",
          "point": "Claude Opus 5.5 · Claude Code",
          "metric": "caching-cost-with-without/With the cache, as recorded",
          "label": "List-price cost of 5-question sessions with and without the cache (calculation) (With the cache, as recorded)",
          "value": 0.255057,
          "unit": "usd",
          "display": "$0.26",
          "n": 15,
          "calculation": true,
          "context": "Claude Opus 5.5 · calculation: 5-turn cached sessions over a fixed ledger"
        },
        {
          "studySlug": "caching-consistency",
          "chartId": "caching-cost-with-without",
          "series": "Without a cache: every input token at the input price",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "caching-cost-with-without/Without a cache: every input token at the input price",
          "label": "List-price cost of 5-question sessions with and without the cache (calculation) (Without a cache: every input token at the input price)",
          "value": 0.269788,
          "unit": "usd",
          "display": "$0.27",
          "n": 15,
          "calculation": true,
          "context": "Claude Sonnet 5.5 · calculation: 5-turn cached sessions over a fixed ledger"
        },
        {
          "studySlug": "caching-consistency",
          "chartId": "caching-cost-with-without",
          "series": "Without a cache: every input token at the input price",
          "point": "Claude Opus 5.5 · Claude Code",
          "metric": "caching-cost-with-without/Without a cache: every input token at the input price",
          "label": "List-price cost of 5-question sessions with and without the cache (calculation) (Without a cache: every input token at the input price)",
          "value": 0.544228,
          "unit": "usd",
          "display": "$0.54",
          "n": 15,
          "calculation": true,
          "context": "Claude Opus 5.5 · calculation: 5-turn cached sessions over a fixed ledger"
        },
        {
          "studySlug": "caching-consistency",
          "chartId": "caching-latency-first-vs-later",
          "series": "Turn 1 (writes the ledger to the cache)",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "caching-latency-first-vs-later/Turn 1 (writes the ledger to the cache)",
          "label": "Time per turn: first turn vs later turns in a cached session (Turn 1 (writes the ledger to the cache))",
          "value": 1.64,
          "unit": "seconds",
          "display": "1.64 s",
          "n": 3,
          "range": [
            1.58,
            1.79
          ],
          "spanKind": "minmax",
          "context": "Claude Sonnet 5.5 · 5-turn cached sessions over a fixed ledger"
        },
        {
          "studySlug": "caching-consistency",
          "chartId": "caching-latency-first-vs-later",
          "series": "Turn 1 (writes the ledger to the cache)",
          "point": "Claude Opus 5.5 · Claude Code",
          "metric": "caching-latency-first-vs-later/Turn 1 (writes the ledger to the cache)",
          "label": "Time per turn: first turn vs later turns in a cached session (Turn 1 (writes the ledger to the cache))",
          "value": 1.9,
          "unit": "seconds",
          "display": "1.90 s",
          "n": 3,
          "range": [
            1.78,
            4.36
          ],
          "spanKind": "minmax",
          "context": "Claude Opus 5.5 · 5-turn cached sessions over a fixed ledger"
        },
        {
          "studySlug": "caching-consistency",
          "chartId": "caching-latency-first-vs-later",
          "series": "Turns 2-5 (read the ledger from the cache)",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "caching-latency-first-vs-later/Turns 2-5 (read the ledger from the cache)",
          "label": "Time per turn: first turn vs later turns in a cached session (Turns 2-5 (read the ledger from the cache))",
          "value": 1.61,
          "unit": "seconds",
          "display": "1.61 s",
          "n": 12,
          "range": [
            1.35,
            5.63
          ],
          "spanKind": "minmax",
          "context": "Claude Sonnet 5.5 · 5-turn cached sessions over a fixed ledger"
        },
        {
          "studySlug": "caching-consistency",
          "chartId": "caching-latency-first-vs-later",
          "series": "Turns 2-5 (read the ledger from the cache)",
          "point": "Claude Opus 5.5 · Claude Code",
          "metric": "caching-latency-first-vs-later/Turns 2-5 (read the ledger from the cache)",
          "label": "Time per turn: first turn vs later turns in a cached session (Turns 2-5 (read the ledger from the cache))",
          "value": 2.4,
          "unit": "seconds",
          "display": "2.40 s",
          "n": 12,
          "range": [
            1.63,
            12.67
          ],
          "spanKind": "minmax",
          "context": "Claude Opus 5.5 · 5-turn cached sessions over a fixed ledger"
        },
        {
          "studySlug": "caching-consistency",
          "chartId": "consistency-pass-rate",
          "series": "Exact number",
          "point": "Claude Haiku 4.5 · Claude Code",
          "metric": "consistency-pass-rate/Exact number",
          "label": "Same prompt, 10 times: strict pass rate (Exact number)",
          "value": 0,
          "unit": "rate",
          "display": "0% (0/10)",
          "n": 10,
          "ci": [
            0,
            0.2775
          ],
          "spanKind": "ci95",
          "context": "Claude Haiku 4.5 · same prompt repeated 10 times"
        },
        {
          "studySlug": "caching-consistency",
          "chartId": "consistency-pass-rate",
          "series": "Exact number",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "consistency-pass-rate/Exact number",
          "label": "Same prompt, 10 times: strict pass rate (Exact number)",
          "value": 1,
          "unit": "rate",
          "display": "100% (10/10)",
          "n": 10,
          "ci": [
            0.7225,
            1
          ],
          "spanKind": "ci95",
          "context": "Claude Sonnet 5.5 · same prompt repeated 10 times"
        },
        {
          "studySlug": "caching-consistency",
          "chartId": "consistency-pass-rate",
          "series": "JSON object",
          "point": "Claude Haiku 4.5 · Claude Code",
          "metric": "consistency-pass-rate/JSON object",
          "label": "Same prompt, 10 times: strict pass rate (JSON object)",
          "value": 0.1,
          "unit": "rate",
          "display": "10% (1/10)",
          "n": 10,
          "ci": [
            0.0179,
            0.4042
          ],
          "spanKind": "ci95",
          "context": "Claude Haiku 4.5 · same prompt repeated 10 times"
        },
        {
          "studySlug": "caching-consistency",
          "chartId": "consistency-pass-rate",
          "series": "JSON object",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "consistency-pass-rate/JSON object",
          "label": "Same prompt, 10 times: strict pass rate (JSON object)",
          "value": 1,
          "unit": "rate",
          "display": "100% (10/10)",
          "n": 10,
          "ci": [
            0.7225,
            1
          ],
          "spanKind": "ci95",
          "context": "Claude Sonnet 5.5 · same prompt repeated 10 times"
        },
        {
          "studySlug": "caching-consistency",
          "chartId": "consistency-pass-rate",
          "series": "Code fix",
          "point": "Claude Haiku 4.5 · Claude Code",
          "metric": "consistency-pass-rate/Code fix",
          "label": "Same prompt, 10 times: strict pass rate (Code fix)",
          "value": 1,
          "unit": "rate",
          "display": "100% (10/10)",
          "n": 10,
          "ci": [
            0.7225,
            1
          ],
          "spanKind": "ci95",
          "context": "Claude Haiku 4.5 · same prompt repeated 10 times"
        },
        {
          "studySlug": "caching-consistency",
          "chartId": "consistency-pass-rate",
          "series": "Code fix",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "consistency-pass-rate/Code fix",
          "label": "Same prompt, 10 times: strict pass rate (Code fix)",
          "value": 1,
          "unit": "rate",
          "display": "100% (10/10)",
          "n": 10,
          "ci": [
            0.7225,
            1
          ],
          "spanKind": "ci95",
          "context": "Claude Sonnet 5.5 · same prompt repeated 10 times"
        },
        {
          "studySlug": "caching-consistency",
          "chartId": "consistency-distinct-answers",
          "series": "Exact number",
          "point": "Claude Haiku 4.5 · Claude Code",
          "metric": "consistency-distinct-answers/Exact number",
          "label": "Same prompt, 10 times: how many different answers (Exact number)",
          "value": 1,
          "unit": "count",
          "display": "1",
          "n": 10,
          "context": "Claude Haiku 4.5 · same prompt repeated 10 times"
        },
        {
          "studySlug": "caching-consistency",
          "chartId": "consistency-distinct-answers",
          "series": "Exact number",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "consistency-distinct-answers/Exact number",
          "label": "Same prompt, 10 times: how many different answers (Exact number)",
          "value": 1,
          "unit": "count",
          "display": "1",
          "n": 10,
          "context": "Claude Sonnet 5.5 · same prompt repeated 10 times"
        },
        {
          "studySlug": "caching-consistency",
          "chartId": "consistency-distinct-answers",
          "series": "JSON object",
          "point": "Claude Haiku 4.5 · Claude Code",
          "metric": "consistency-distinct-answers/JSON object",
          "label": "Same prompt, 10 times: how many different answers (JSON object)",
          "value": 1,
          "unit": "count",
          "display": "1",
          "n": 10,
          "context": "Claude Haiku 4.5 · same prompt repeated 10 times"
        },
        {
          "studySlug": "caching-consistency",
          "chartId": "consistency-distinct-answers",
          "series": "JSON object",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "consistency-distinct-answers/JSON object",
          "label": "Same prompt, 10 times: how many different answers (JSON object)",
          "value": 1,
          "unit": "count",
          "display": "1",
          "n": 10,
          "context": "Claude Sonnet 5.5 · same prompt repeated 10 times"
        },
        {
          "studySlug": "caching-consistency",
          "chartId": "consistency-distinct-answers",
          "series": "Code fix",
          "point": "Claude Haiku 4.5 · Claude Code",
          "metric": "consistency-distinct-answers/Code fix",
          "label": "Same prompt, 10 times: how many different answers (Code fix)",
          "value": 6,
          "unit": "count",
          "display": "6",
          "n": 10,
          "context": "Claude Haiku 4.5 · same prompt repeated 10 times"
        },
        {
          "studySlug": "caching-consistency",
          "chartId": "consistency-distinct-answers",
          "series": "Code fix",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "consistency-distinct-answers/Code fix",
          "label": "Same prompt, 10 times: how many different answers (Code fix)",
          "value": 3,
          "unit": "count",
          "display": "3",
          "n": 10,
          "context": "Claude Sonnet 5.5 · same prompt repeated 10 times"
        },
        {
          "studySlug": "caching-consistency",
          "chartId": "consistency-latency-spread",
          "series": "Exact number",
          "point": "Claude Haiku 4.5 · Claude Code",
          "metric": "consistency-latency-spread/Exact number",
          "label": "Same prompt, 10 times: time per call (Exact number)",
          "value": 5.06,
          "unit": "seconds",
          "display": "5.06 s",
          "n": 10,
          "range": [
            4.42,
            6.2
          ],
          "spanKind": "minmax",
          "context": "Claude Haiku 4.5 · same prompt repeated 10 times"
        },
        {
          "studySlug": "caching-consistency",
          "chartId": "consistency-latency-spread",
          "series": "Exact number",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "consistency-latency-spread/Exact number",
          "label": "Same prompt, 10 times: time per call (Exact number)",
          "value": 6.89,
          "unit": "seconds",
          "display": "6.89 s",
          "n": 10,
          "range": [
            5.81,
            7.81
          ],
          "spanKind": "minmax",
          "context": "Claude Sonnet 5.5 · same prompt repeated 10 times"
        },
        {
          "studySlug": "caching-consistency",
          "chartId": "consistency-latency-spread",
          "series": "JSON object",
          "point": "Claude Haiku 4.5 · Claude Code",
          "metric": "consistency-latency-spread/JSON object",
          "label": "Same prompt, 10 times: time per call (JSON object)",
          "value": 7.03,
          "unit": "seconds",
          "display": "7.03 s",
          "n": 10,
          "range": [
            5.28,
            12.27
          ],
          "spanKind": "minmax",
          "context": "Claude Haiku 4.5 · same prompt repeated 10 times"
        },
        {
          "studySlug": "caching-consistency",
          "chartId": "consistency-latency-spread",
          "series": "JSON object",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "consistency-latency-spread/JSON object",
          "label": "Same prompt, 10 times: time per call (JSON object)",
          "value": 2.89,
          "unit": "seconds",
          "display": "2.89 s",
          "n": 10,
          "range": [
            2.68,
            5.3
          ],
          "spanKind": "minmax",
          "context": "Claude Sonnet 5.5 · same prompt repeated 10 times"
        },
        {
          "studySlug": "caching-consistency",
          "chartId": "consistency-latency-spread",
          "series": "Code fix",
          "point": "Claude Haiku 4.5 · Claude Code",
          "metric": "consistency-latency-spread/Code fix",
          "label": "Same prompt, 10 times: time per call (Code fix)",
          "value": 5.95,
          "unit": "seconds",
          "display": "5.95 s",
          "n": 10,
          "range": [
            4.89,
            7.33
          ],
          "spanKind": "minmax",
          "context": "Claude Haiku 4.5 · same prompt repeated 10 times"
        },
        {
          "studySlug": "caching-consistency",
          "chartId": "consistency-latency-spread",
          "series": "Code fix",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "consistency-latency-spread/Code fix",
          "label": "Same prompt, 10 times: time per call (Code fix)",
          "value": 2.67,
          "unit": "seconds",
          "display": "2.67 s",
          "n": 10,
          "range": [
            2.32,
            4.34
          ],
          "spanKind": "minmax",
          "context": "Claude Sonnet 5.5 · same prompt repeated 10 times"
        },
        {
          "studySlug": "agent-memory",
          "statId": "memory-sessions",
          "metric": "stat:memory-sessions",
          "label": "Claude Code sessions, every one graded (120 Sonnet 5.5, 80 Haiku 4.5)",
          "value": 200,
          "unit": "count",
          "display": "200",
          "n": 200,
          "context": ""
        },
        {
          "studySlug": "routing-overhead",
          "chartId": "cli-startup-tax",
          "series": "First output event",
          "point": "Claude Code · Claude Haiku 4.5",
          "metric": "cli-startup-tax/First output event",
          "label": "CLI start-up tax on a one-word answer (First output event)",
          "value": 563,
          "unit": "ms",
          "display": "563 ms",
          "n": 5,
          "range": [
            519,
            726
          ],
          "spanKind": "minmax",
          "context": "Claude Haiku 4.5 · CLI start-up, one-word prompt, 5 runs"
        },
        {
          "studySlug": "routing-overhead",
          "chartId": "cli-startup-tax",
          "series": "First model output",
          "point": "Claude Code · Claude Haiku 4.5",
          "metric": "cli-startup-tax/First model output",
          "label": "CLI start-up tax on a one-word answer (First model output)",
          "value": 1461,
          "unit": "ms",
          "display": "1,461 ms",
          "n": 5,
          "range": [
            1206,
            2308
          ],
          "spanKind": "minmax",
          "context": "Claude Haiku 4.5 · CLI start-up, one-word prompt, 5 runs"
        },
        {
          "studySlug": "routing-overhead",
          "chartId": "cli-startup-tax",
          "series": "Total wall time",
          "point": "Claude Code · Claude Haiku 4.5",
          "metric": "cli-startup-tax/Total wall time",
          "label": "CLI start-up tax on a one-word answer (Total wall time)",
          "value": 2529,
          "unit": "ms",
          "display": "2,529 ms",
          "n": 5,
          "range": [
            2273,
            3382
          ],
          "spanKind": "minmax",
          "context": "Claude Haiku 4.5 · CLI start-up, one-word prompt, 5 runs"
        },
        {
          "studySlug": "routing-overhead",
          "chartId": "cli-startup-input-tokens",
          "series": "Input tokens per call",
          "point": "Claude Code · Claude Haiku 4.5",
          "metric": "cli-startup-input-tokens",
          "label": "Input tokens a CLI sends for a one-word answer",
          "value": 6761,
          "unit": "tokens",
          "display": "6,761",
          "n": 5,
          "context": "Claude Haiku 4.5 · CLI start-up, one-word prompt, 5 runs"
        },
        {
          "studySlug": "routing-overhead",
          "statId": "cli-startup-claude-harness-ms",
          "metric": "stat:cli-startup-claude-harness-ms",
          "label": "Claude Code time outside the model on a one-word answer",
          "value": 1690,
          "unit": "ms",
          "display": "1,690 ms median (1,533 to 1,811)",
          "n": 5,
          "context": "routing overhead per decision"
        },
        {
          "studySlug": "cli-model-latency-tokens",
          "chartId": "scheduler-repair-claude-vs-codex",
          "series": "Total time",
          "point": "Claude Code CLI · Sonnet 5.5 · medium",
          "metric": "scheduler-repair-claude-vs-codex/Total time",
          "label": "Repairing a scheduler: Claude Code vs Codex vs API (Total time)",
          "value": 15,
          "unit": "seconds",
          "display": "15.0 s",
          "n": 3,
          "range": [
            13.89,
            15.89
          ],
          "spanKind": "minmax",
          "context": "Sonnet 5.5 · effort medium · scheduler repair, 296 checks, 3 runs"
        },
        {
          "studySlug": "cli-model-latency-tokens",
          "chartId": "scheduler-repair-claude-vs-codex",
          "series": "First useful output",
          "point": "Claude Code CLI · Sonnet 5.5 · medium",
          "metric": "scheduler-repair-claude-vs-codex/First useful output",
          "label": "Repairing a scheduler: Claude Code vs Codex vs API (First useful output)",
          "value": 7.55,
          "unit": "seconds",
          "display": "7.55 s",
          "n": 3,
          "range": [
            6.77,
            7.63
          ],
          "spanKind": "minmax",
          "context": "Sonnet 5.5 · effort medium · scheduler repair, 296 checks, 3 runs"
        },
        {
          "studySlug": "cli-model-latency-tokens",
          "chartId": "scheduler-repair-output-tokens",
          "series": "Output tokens",
          "point": "Claude Code CLI · Sonnet 5.5 · medium",
          "metric": "scheduler-repair-output-tokens/Output tokens",
          "label": "Output tokens to repair the scheduler (Output tokens)",
          "value": 2227,
          "unit": "tokens",
          "display": "2,227",
          "n": 3,
          "context": "Sonnet 5.5 · effort medium · scheduler repair, 296 checks, 3 runs"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-pass-rate",
          "series": "Strict pass",
          "point": "Claude Haiku 4.5 (single call) · Claude Code",
          "metric": "agent-loop-pass-rate",
          "label": "Strict pass rate: single call vs agent loop on eight hard tasks",
          "value": 0.4583,
          "unit": "rate",
          "display": "46% (11/24)",
          "n": 24,
          "ci": [
            0.2789,
            0.6493
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Haiku 4.5 · single call"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-pass-rate",
          "series": "Strict pass",
          "point": "Claude Haiku 4.5 (agent loop) · Claude Code",
          "metric": "agent-loop-pass-rate",
          "label": "Strict pass rate: single call vs agent loop on eight hard tasks",
          "value": 0.5417,
          "unit": "rate",
          "display": "54% (13/24)",
          "n": 24,
          "ci": [
            0.3507,
            0.7211
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Haiku 4.5 · agent loop"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-pass-rate",
          "series": "Strict pass",
          "point": "Claude Sonnet 5.5 (single call) · Claude Code",
          "metric": "agent-loop-pass-rate",
          "label": "Strict pass rate: single call vs agent loop on eight hard tasks",
          "value": 1,
          "unit": "rate",
          "display": "100% (24/24)",
          "n": 24,
          "ci": [
            0.862,
            1
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Sonnet 5.5 · single call"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-pass-rate",
          "series": "Strict pass",
          "point": "Claude Sonnet 5.5 (agent loop) · Claude Code",
          "metric": "agent-loop-pass-rate",
          "label": "Strict pass rate: single call vs agent loop on eight hard tasks",
          "value": 1,
          "unit": "rate",
          "display": "100% (16/16)",
          "n": 16,
          "ci": [
            0.8064,
            1
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Sonnet 5.5 · agent loop"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "series": "Claude Haiku 4.5 (single call) · Claude Code",
          "point": "Interval merge fix",
          "metric": "agent-loop-by-task/Interval merge fix",
          "label": "Strict passes per task: single call vs agent loop: Interval merge fix",
          "value": 1,
          "unit": "rate",
          "display": "100% (3/3)",
          "n": 3,
          "ci": [
            0.4385,
            1
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Haiku 4.5 · single call"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "series": "Claude Haiku 4.5 (single call) · Claude Code",
          "point": "DST day-length fix",
          "metric": "agent-loop-by-task/DST day-length fix",
          "label": "Strict passes per task: single call vs agent loop: DST day-length fix",
          "value": 0.3333,
          "unit": "rate",
          "display": "33% (1/3)",
          "n": 3,
          "ci": [
            0.0615,
            0.7923
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Haiku 4.5 · single call"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "series": "Claude Haiku 4.5 (single call) · Claude Code",
          "point": "CSV parser",
          "metric": "agent-loop-by-task/CSV parser",
          "label": "Strict passes per task: single call vs agent loop: CSV parser",
          "value": 0.6667,
          "unit": "rate",
          "display": "67% (2/3)",
          "n": 3,
          "ci": [
            0.2077,
            0.9385
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Haiku 4.5 · single call"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "series": "Claude Haiku 4.5 (single call) · Claude Code",
          "point": "Event-loop order",
          "metric": "agent-loop-by-task/Event-loop order",
          "label": "Strict passes per task: single call vs agent loop: Event-loop order",
          "value": 0,
          "unit": "rate",
          "display": "0% (0/3)",
          "n": 3,
          "ci": [
            0,
            0.5615
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Haiku 4.5 · single call"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "series": "Claude Haiku 4.5 (single call) · Claude Code",
          "point": "Room schedule",
          "metric": "agent-loop-by-task/Room schedule",
          "label": "Strict passes per task: single call vs agent loop: Room schedule",
          "value": 0,
          "unit": "rate",
          "display": "0% (0/3)",
          "n": 3,
          "ci": [
            0,
            0.5615
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Haiku 4.5 · single call"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "series": "Claude Haiku 4.5 (single call) · Claude Code",
          "point": "SemVer regex",
          "metric": "agent-loop-by-task/SemVer regex",
          "label": "Strict passes per task: single call vs agent loop: SemVer regex",
          "value": 1,
          "unit": "rate",
          "display": "100% (3/3)",
          "n": 3,
          "ci": [
            0.4385,
            1
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Haiku 4.5 · single call"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "series": "Claude Haiku 4.5 (single call) · Claude Code",
          "point": "Money refactor",
          "metric": "agent-loop-by-task/Money refactor",
          "label": "Strict passes per task: single call vs agent loop: Money refactor",
          "value": 0.6667,
          "unit": "rate",
          "display": "67% (2/3)",
          "n": 3,
          "ci": [
            0.2077,
            0.9385
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Haiku 4.5 · single call"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "series": "Claude Haiku 4.5 (single call) · Claude Code",
          "point": "SQL report",
          "metric": "agent-loop-by-task/SQL report",
          "label": "Strict passes per task: single call vs agent loop: SQL report",
          "value": 0,
          "unit": "rate",
          "display": "0% (0/3)",
          "n": 3,
          "ci": [
            0,
            0.5615
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Haiku 4.5 · single call"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "series": "Claude Haiku 4.5 (agent loop) · Claude Code",
          "point": "Interval merge fix",
          "metric": "agent-loop-by-task/Interval merge fix",
          "label": "Strict passes per task: single call vs agent loop: Interval merge fix",
          "value": 0.6667,
          "unit": "rate",
          "display": "67% (2/3)",
          "n": 3,
          "ci": [
            0.2077,
            0.9385
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Haiku 4.5 · agent loop"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "series": "Claude Haiku 4.5 (agent loop) · Claude Code",
          "point": "DST day-length fix",
          "metric": "agent-loop-by-task/DST day-length fix",
          "label": "Strict passes per task: single call vs agent loop: DST day-length fix",
          "value": 0.6667,
          "unit": "rate",
          "display": "67% (2/3)",
          "n": 3,
          "ci": [
            0.2077,
            0.9385
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Haiku 4.5 · agent loop"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "series": "Claude Haiku 4.5 (agent loop) · Claude Code",
          "point": "CSV parser",
          "metric": "agent-loop-by-task/CSV parser",
          "label": "Strict passes per task: single call vs agent loop: CSV parser",
          "value": 0.6667,
          "unit": "rate",
          "display": "67% (2/3)",
          "n": 3,
          "ci": [
            0.2077,
            0.9385
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Haiku 4.5 · agent loop"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "series": "Claude Haiku 4.5 (agent loop) · Claude Code",
          "point": "Event-loop order",
          "metric": "agent-loop-by-task/Event-loop order",
          "label": "Strict passes per task: single call vs agent loop: Event-loop order",
          "value": 1,
          "unit": "rate",
          "display": "100% (3/3)",
          "n": 3,
          "ci": [
            0.4385,
            1
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Haiku 4.5 · agent loop"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "series": "Claude Haiku 4.5 (agent loop) · Claude Code",
          "point": "Room schedule",
          "metric": "agent-loop-by-task/Room schedule",
          "label": "Strict passes per task: single call vs agent loop: Room schedule",
          "value": 0.6667,
          "unit": "rate",
          "display": "67% (2/3)",
          "n": 3,
          "ci": [
            0.2077,
            0.9385
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Haiku 4.5 · agent loop"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "series": "Claude Haiku 4.5 (agent loop) · Claude Code",
          "point": "SemVer regex",
          "metric": "agent-loop-by-task/SemVer regex",
          "label": "Strict passes per task: single call vs agent loop: SemVer regex",
          "value": 0.6667,
          "unit": "rate",
          "display": "67% (2/3)",
          "n": 3,
          "ci": [
            0.2077,
            0.9385
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Haiku 4.5 · agent loop"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "series": "Claude Haiku 4.5 (agent loop) · Claude Code",
          "point": "Money refactor",
          "metric": "agent-loop-by-task/Money refactor",
          "label": "Strict passes per task: single call vs agent loop: Money refactor",
          "value": 0,
          "unit": "rate",
          "display": "0% (0/3)",
          "n": 3,
          "ci": [
            0,
            0.5615
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Haiku 4.5 · agent loop"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "series": "Claude Haiku 4.5 (agent loop) · Claude Code",
          "point": "SQL report",
          "metric": "agent-loop-by-task/SQL report",
          "label": "Strict passes per task: single call vs agent loop: SQL report",
          "value": 0,
          "unit": "rate",
          "display": "0% (0/3)",
          "n": 3,
          "ci": [
            0,
            0.5615
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Haiku 4.5 · agent loop"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "series": "Claude Sonnet 5.5 (single call) · Claude Code",
          "point": "Interval merge fix",
          "metric": "agent-loop-by-task/Interval merge fix",
          "label": "Strict passes per task: single call vs agent loop: Interval merge fix",
          "value": 1,
          "unit": "rate",
          "display": "100% (3/3)",
          "n": 3,
          "ci": [
            0.4385,
            1
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Sonnet 5.5 · single call"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "series": "Claude Sonnet 5.5 (single call) · Claude Code",
          "point": "DST day-length fix",
          "metric": "agent-loop-by-task/DST day-length fix",
          "label": "Strict passes per task: single call vs agent loop: DST day-length fix",
          "value": 1,
          "unit": "rate",
          "display": "100% (3/3)",
          "n": 3,
          "ci": [
            0.4385,
            1
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Sonnet 5.5 · single call"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "series": "Claude Sonnet 5.5 (single call) · Claude Code",
          "point": "CSV parser",
          "metric": "agent-loop-by-task/CSV parser",
          "label": "Strict passes per task: single call vs agent loop: CSV parser",
          "value": 1,
          "unit": "rate",
          "display": "100% (3/3)",
          "n": 3,
          "ci": [
            0.4385,
            1
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Sonnet 5.5 · single call"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "series": "Claude Sonnet 5.5 (single call) · Claude Code",
          "point": "Event-loop order",
          "metric": "agent-loop-by-task/Event-loop order",
          "label": "Strict passes per task: single call vs agent loop: Event-loop order",
          "value": 1,
          "unit": "rate",
          "display": "100% (3/3)",
          "n": 3,
          "ci": [
            0.4385,
            1
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Sonnet 5.5 · single call"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "series": "Claude Sonnet 5.5 (single call) · Claude Code",
          "point": "Room schedule",
          "metric": "agent-loop-by-task/Room schedule",
          "label": "Strict passes per task: single call vs agent loop: Room schedule",
          "value": 1,
          "unit": "rate",
          "display": "100% (3/3)",
          "n": 3,
          "ci": [
            0.4385,
            1
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Sonnet 5.5 · single call"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "series": "Claude Sonnet 5.5 (single call) · Claude Code",
          "point": "SemVer regex",
          "metric": "agent-loop-by-task/SemVer regex",
          "label": "Strict passes per task: single call vs agent loop: SemVer regex",
          "value": 1,
          "unit": "rate",
          "display": "100% (3/3)",
          "n": 3,
          "ci": [
            0.4385,
            1
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Sonnet 5.5 · single call"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "series": "Claude Sonnet 5.5 (single call) · Claude Code",
          "point": "Money refactor",
          "metric": "agent-loop-by-task/Money refactor",
          "label": "Strict passes per task: single call vs agent loop: Money refactor",
          "value": 1,
          "unit": "rate",
          "display": "100% (3/3)",
          "n": 3,
          "ci": [
            0.4385,
            1
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Sonnet 5.5 · single call"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "series": "Claude Sonnet 5.5 (single call) · Claude Code",
          "point": "SQL report",
          "metric": "agent-loop-by-task/SQL report",
          "label": "Strict passes per task: single call vs agent loop: SQL report",
          "value": 1,
          "unit": "rate",
          "display": "100% (3/3)",
          "n": 3,
          "ci": [
            0.4385,
            1
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Sonnet 5.5 · single call"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "series": "Claude Sonnet 5.5 (agent loop) · Claude Code",
          "point": "Interval merge fix",
          "metric": "agent-loop-by-task/Interval merge fix",
          "label": "Strict passes per task: single call vs agent loop: Interval merge fix",
          "value": 1,
          "unit": "rate",
          "display": "100% (2/2)",
          "n": 2,
          "ci": [
            0.3424,
            1
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Sonnet 5.5 · agent loop"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "series": "Claude Sonnet 5.5 (agent loop) · Claude Code",
          "point": "DST day-length fix",
          "metric": "agent-loop-by-task/DST day-length fix",
          "label": "Strict passes per task: single call vs agent loop: DST day-length fix",
          "value": 1,
          "unit": "rate",
          "display": "100% (2/2)",
          "n": 2,
          "ci": [
            0.3424,
            1
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Sonnet 5.5 · agent loop"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "series": "Claude Sonnet 5.5 (agent loop) · Claude Code",
          "point": "CSV parser",
          "metric": "agent-loop-by-task/CSV parser",
          "label": "Strict passes per task: single call vs agent loop: CSV parser",
          "value": 1,
          "unit": "rate",
          "display": "100% (2/2)",
          "n": 2,
          "ci": [
            0.3424,
            1
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Sonnet 5.5 · agent loop"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "series": "Claude Sonnet 5.5 (agent loop) · Claude Code",
          "point": "Event-loop order",
          "metric": "agent-loop-by-task/Event-loop order",
          "label": "Strict passes per task: single call vs agent loop: Event-loop order",
          "value": 1,
          "unit": "rate",
          "display": "100% (2/2)",
          "n": 2,
          "ci": [
            0.3424,
            1
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Sonnet 5.5 · agent loop"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "series": "Claude Sonnet 5.5 (agent loop) · Claude Code",
          "point": "Room schedule",
          "metric": "agent-loop-by-task/Room schedule",
          "label": "Strict passes per task: single call vs agent loop: Room schedule",
          "value": 1,
          "unit": "rate",
          "display": "100% (2/2)",
          "n": 2,
          "ci": [
            0.3424,
            1
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Sonnet 5.5 · agent loop"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "series": "Claude Sonnet 5.5 (agent loop) · Claude Code",
          "point": "SemVer regex",
          "metric": "agent-loop-by-task/SemVer regex",
          "label": "Strict passes per task: single call vs agent loop: SemVer regex",
          "value": 1,
          "unit": "rate",
          "display": "100% (2/2)",
          "n": 2,
          "ci": [
            0.3424,
            1
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Sonnet 5.5 · agent loop"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "series": "Claude Sonnet 5.5 (agent loop) · Claude Code",
          "point": "Money refactor",
          "metric": "agent-loop-by-task/Money refactor",
          "label": "Strict passes per task: single call vs agent loop: Money refactor",
          "value": 1,
          "unit": "rate",
          "display": "100% (2/2)",
          "n": 2,
          "ci": [
            0.3424,
            1
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Sonnet 5.5 · agent loop"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "series": "Claude Sonnet 5.5 (agent loop) · Claude Code",
          "point": "SQL report",
          "metric": "agent-loop-by-task/SQL report",
          "label": "Strict passes per task: single call vs agent loop: SQL report",
          "value": 1,
          "unit": "rate",
          "display": "100% (2/2)",
          "n": 2,
          "ci": [
            0.3424,
            1
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Sonnet 5.5 · agent loop"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-total-time",
          "series": "Total time per attempt",
          "point": "Claude Haiku 4.5 (single call) · Claude Code",
          "metric": "agent-loop-total-time",
          "label": "Total time per attempt: single call vs agent loop",
          "value": 39.01,
          "unit": "seconds",
          "display": "39.0 s",
          "n": 24,
          "range": [
            15.27,
            75.13
          ],
          "spanKind": "minmax",
          "context": "Claude Haiku 4.5 · single call"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-total-time",
          "series": "Total time per attempt",
          "point": "Claude Haiku 4.5 (agent loop) · Claude Code",
          "metric": "agent-loop-total-time",
          "label": "Total time per attempt: single call vs agent loop",
          "value": 56.77,
          "unit": "seconds",
          "display": "56.8 s",
          "n": 24,
          "range": [
            24.53,
            223.7
          ],
          "spanKind": "minmax",
          "context": "Claude Haiku 4.5 · agent loop"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-total-time",
          "series": "Total time per attempt",
          "point": "Claude Sonnet 5.5 (single call) · Claude Code",
          "metric": "agent-loop-total-time",
          "label": "Total time per attempt: single call vs agent loop",
          "value": 7.75,
          "unit": "seconds",
          "display": "7.75 s",
          "n": 24,
          "range": [
            2.26,
            34.79
          ],
          "spanKind": "minmax",
          "context": "Claude Sonnet 5.5 · single call"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-total-time",
          "series": "Total time per attempt",
          "point": "Claude Sonnet 5.5 (agent loop) · Claude Code",
          "metric": "agent-loop-total-time",
          "label": "Total time per attempt: single call vs agent loop",
          "value": 7.41,
          "unit": "seconds",
          "display": "7.41 s",
          "n": 16,
          "range": [
            2.75,
            24.19
          ],
          "spanKind": "minmax",
          "context": "Claude Sonnet 5.5 · agent loop"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-tokens",
          "series": "Input tokens (cache reads included)",
          "point": "Claude Haiku 4.5 (single call) · Claude Code",
          "metric": "agent-loop-tokens/Input tokens (cache reads included)",
          "label": "Tokens per attempt: single call vs agent loop (Input tokens (cache reads included))",
          "value": 3941,
          "unit": "tokens",
          "display": "3,941",
          "n": 24,
          "range": [
            3879,
            4221
          ],
          "spanKind": "minmax",
          "context": "Claude Haiku 4.5 · single call"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-tokens",
          "series": "Input tokens (cache reads included)",
          "point": "Claude Haiku 4.5 (agent loop) · Claude Code",
          "metric": "agent-loop-tokens/Input tokens (cache reads included)",
          "label": "Tokens per attempt: single call vs agent loop (Input tokens (cache reads included))",
          "value": 71691,
          "unit": "tokens",
          "display": "71,691",
          "n": 24,
          "range": [
            41732,
            516306
          ],
          "spanKind": "minmax",
          "context": "Claude Haiku 4.5 · agent loop"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-tokens",
          "series": "Input tokens (cache reads included)",
          "point": "Claude Sonnet 5.5 (single call) · Claude Code",
          "metric": "agent-loop-tokens/Input tokens (cache reads included)",
          "label": "Tokens per attempt: single call vs agent loop (Input tokens (cache reads included))",
          "value": 2281,
          "unit": "tokens",
          "display": "2,281",
          "n": 24,
          "range": [
            2234,
            2669
          ],
          "spanKind": "minmax",
          "context": "Claude Sonnet 5.5 · single call"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-tokens",
          "series": "Input tokens (cache reads included)",
          "point": "Claude Sonnet 5.5 (agent loop) · Claude Code",
          "metric": "agent-loop-tokens/Input tokens (cache reads included)",
          "label": "Tokens per attempt: single call vs agent loop (Input tokens (cache reads included))",
          "value": 9550,
          "unit": "tokens",
          "display": "9,550",
          "n": 16,
          "range": [
            9398,
            33040
          ],
          "spanKind": "minmax",
          "context": "Claude Sonnet 5.5 · agent loop"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-tokens",
          "series": "Output tokens",
          "point": "Claude Haiku 4.5 (single call) · Claude Code",
          "metric": "agent-loop-tokens/Output tokens",
          "label": "Tokens per attempt: single call vs agent loop (Output tokens)",
          "value": 5064,
          "unit": "tokens",
          "display": "5,064",
          "n": 24,
          "range": [
            1899,
            9321
          ],
          "spanKind": "minmax",
          "context": "Claude Haiku 4.5 · single call"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-tokens",
          "series": "Output tokens",
          "point": "Claude Haiku 4.5 (agent loop) · Claude Code",
          "metric": "agent-loop-tokens/Output tokens",
          "label": "Tokens per attempt: single call vs agent loop (Output tokens)",
          "value": 7912,
          "unit": "tokens",
          "display": "7,912",
          "n": 24,
          "range": [
            2541,
            20654
          ],
          "spanKind": "minmax",
          "context": "Claude Haiku 4.5 · agent loop"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-tokens",
          "series": "Output tokens",
          "point": "Claude Sonnet 5.5 (single call) · Claude Code",
          "metric": "agent-loop-tokens/Output tokens",
          "label": "Tokens per attempt: single call vs agent loop (Output tokens)",
          "value": 1050,
          "unit": "tokens",
          "display": "1,050",
          "n": 24,
          "range": [
            176,
            3895
          ],
          "spanKind": "minmax",
          "context": "Claude Sonnet 5.5 · single call"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-tokens",
          "series": "Output tokens",
          "point": "Claude Sonnet 5.5 (agent loop) · Claude Code",
          "metric": "agent-loop-tokens/Output tokens",
          "label": "Tokens per attempt: single call vs agent loop (Output tokens)",
          "value": 876,
          "unit": "tokens",
          "display": "876",
          "n": 16,
          "range": [
            219,
            3243
          ],
          "spanKind": "minmax",
          "context": "Claude Sonnet 5.5 · agent loop"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-tool-calls",
          "series": "Tool calls per attempt",
          "point": "Claude Haiku 4.5 (agent loop) · Claude Code",
          "metric": "agent-loop-tool-calls",
          "label": "Tool calls per agent-loop attempt",
          "value": 3,
          "unit": "count",
          "display": "3",
          "n": 24,
          "range": [
            2,
            18
          ],
          "spanKind": "minmax",
          "context": "Claude Haiku 4.5 · agent loop"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-tool-calls",
          "series": "Tool calls per attempt",
          "point": "Claude Sonnet 5.5 (agent loop) · Claude Code",
          "metric": "agent-loop-tool-calls",
          "label": "Tool calls per agent-loop attempt",
          "value": 0,
          "unit": "count",
          "display": "0",
          "n": 16,
          "range": [
            0,
            3
          ],
          "spanKind": "minmax",
          "context": "Claude Sonnet 5.5 · agent loop"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-cost-per-pass",
          "series": "Cost per strict pass",
          "point": "Claude Haiku 4.5 (single call) · Claude Code",
          "metric": "agent-loop-cost-per-pass",
          "label": "List-price cost per strict pass: single call vs agent loop (calculation)",
          "value": 0.0672,
          "unit": "usd",
          "display": "$0.067",
          "n": 24,
          "calculation": true,
          "context": "Claude Haiku 4.5 · single call"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-cost-per-pass",
          "series": "Cost per strict pass",
          "point": "Claude Haiku 4.5 (agent loop) · Claude Code",
          "metric": "agent-loop-cost-per-pass",
          "label": "List-price cost per strict pass: single call vs agent loop (calculation)",
          "value": 0.14225,
          "unit": "usd",
          "display": "$0.14",
          "n": 24,
          "calculation": true,
          "context": "Claude Haiku 4.5 · agent loop"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-cost-per-pass",
          "series": "Cost per strict pass",
          "point": "Claude Sonnet 5.5 (single call) · Claude Code",
          "metric": "agent-loop-cost-per-pass",
          "label": "List-price cost per strict pass: single call vs agent loop (calculation)",
          "value": 0.01435,
          "unit": "usd",
          "display": "$0.014",
          "n": 24,
          "calculation": true,
          "context": "Claude Sonnet 5.5 · single call"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-cost-per-pass",
          "series": "Cost per strict pass",
          "point": "Claude Sonnet 5.5 (agent loop) · Claude Code",
          "metric": "agent-loop-cost-per-pass",
          "label": "List-price cost per strict pass: single call vs agent loop (calculation)",
          "value": 0.02746,
          "unit": "usd",
          "display": "$0.027",
          "n": 16,
          "calculation": true,
          "context": "Claude Sonnet 5.5 · agent loop"
        },
        {
          "studySlug": "haiku-thinking-on-off",
          "chartId": "haiku-thinking-router-exact",
          "series": "Exact decisions (every scored question right)",
          "point": "Claude Haiku 4.5 (thinking off) · Claude Code",
          "metric": "haiku-thinking-router-exact/Exact decisions (every scored question right)",
          "label": "Haiku thinking study: typed routing decisions answered exactly right (Exact decisions (every scored question right))",
          "value": 0.8659,
          "unit": "rate",
          "display": "87% (71/82)",
          "n": 82,
          "ci": [
            0.7755,
            0.9234
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Haiku 4.5 · thinking off · typed routing decisions, thinking on vs off"
        },
        {
          "studySlug": "haiku-thinking-on-off",
          "chartId": "haiku-thinking-router-exact",
          "series": "Exact decisions (every scored question right)",
          "point": "Claude Haiku 4.5 (thinking on) · Claude Code",
          "metric": "haiku-thinking-router-exact/Exact decisions (every scored question right)",
          "label": "Haiku thinking study: typed routing decisions answered exactly right (Exact decisions (every scored question right))",
          "value": 0.8902,
          "unit": "rate",
          "display": "89% (73/82)",
          "n": 82,
          "ci": [
            0.8044,
            0.9412
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Haiku 4.5 · thinking on · typed routing decisions, thinking on vs off"
        },
        {
          "studySlug": "haiku-thinking-on-off",
          "chartId": "haiku-thinking-router-exact",
          "series": "Exact decisions (every scored question right)",
          "point": "Claude Sonnet 5.5 (low) · Claude Code",
          "metric": "haiku-thinking-router-exact/Exact decisions (every scored question right)",
          "label": "Haiku thinking study: typed routing decisions answered exactly right (Exact decisions (every scored question right))",
          "value": 0.939,
          "unit": "rate",
          "display": "94% (77/82)",
          "n": 82,
          "ci": [
            0.8651,
            0.9737
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Sonnet 5.5 · effort low · typed routing decisions, thinking on vs off"
        },
        {
          "studySlug": "haiku-thinking-on-off",
          "chartId": "haiku-thinking-router-exact",
          "series": "Per-question accuracy",
          "point": "Claude Haiku 4.5 (thinking off) · Claude Code",
          "metric": "haiku-thinking-router-exact/Per-question accuracy",
          "label": "Haiku thinking study: typed routing decisions answered exactly right (Per-question accuracy)",
          "value": 0.9124,
          "unit": "rate",
          "display": "91% (177/194)",
          "n": 194,
          "ci": [
            0.8642,
            0.9446
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Haiku 4.5 · thinking off · typed routing decisions, thinking on vs off"
        },
        {
          "studySlug": "haiku-thinking-on-off",
          "chartId": "haiku-thinking-router-exact",
          "series": "Per-question accuracy",
          "point": "Claude Haiku 4.5 (thinking on) · Claude Code",
          "metric": "haiku-thinking-router-exact/Per-question accuracy",
          "label": "Haiku thinking study: typed routing decisions answered exactly right (Per-question accuracy)",
          "value": 0.9433,
          "unit": "rate",
          "display": "94% (183/194)",
          "n": 194,
          "ci": [
            0.9013,
            0.968
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Haiku 4.5 · thinking on · typed routing decisions, thinking on vs off"
        },
        {
          "studySlug": "haiku-thinking-on-off",
          "chartId": "haiku-thinking-router-exact",
          "series": "Per-question accuracy",
          "point": "Claude Sonnet 5.5 (low) · Claude Code",
          "metric": "haiku-thinking-router-exact/Per-question accuracy",
          "label": "Haiku thinking study: typed routing decisions answered exactly right (Per-question accuracy)",
          "value": 0.9742,
          "unit": "rate",
          "display": "97% (189/194)",
          "n": 194,
          "ci": [
            0.9411,
            0.9889
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Sonnet 5.5 · effort low · typed routing decisions, thinking on vs off"
        },
        {
          "studySlug": "haiku-thinking-on-off",
          "chartId": "haiku-thinking-router-latency",
          "series": "Wall time (CLI)",
          "point": "Claude Haiku 4.5 (thinking off) · Claude Code",
          "metric": "haiku-thinking-router-latency/Wall time (CLI)",
          "label": "Haiku thinking study: time per routing decision (Wall time (CLI))",
          "value": 4.66,
          "unit": "seconds",
          "display": "4.66 s",
          "n": 82,
          "range": [
            4.66,
            8.18
          ],
          "spanKind": "p50-p95",
          "context": "Claude Haiku 4.5 · thinking off · typed routing decisions, thinking on vs off"
        },
        {
          "studySlug": "haiku-thinking-on-off",
          "chartId": "haiku-thinking-router-latency",
          "series": "Wall time (CLI)",
          "point": "Claude Haiku 4.5 (thinking on) · Claude Code",
          "metric": "haiku-thinking-router-latency/Wall time (CLI)",
          "label": "Haiku thinking study: time per routing decision (Wall time (CLI))",
          "value": 12.54,
          "unit": "seconds",
          "display": "12.5 s",
          "n": 82,
          "range": [
            12.54,
            34.48
          ],
          "spanKind": "p50-p95",
          "context": "Claude Haiku 4.5 · thinking on · typed routing decisions, thinking on vs off"
        },
        {
          "studySlug": "haiku-thinking-on-off",
          "chartId": "haiku-thinking-router-latency",
          "series": "Wall time (CLI)",
          "point": "Claude Sonnet 5.5 (low) · Claude Code",
          "metric": "haiku-thinking-router-latency/Wall time (CLI)",
          "label": "Haiku thinking study: time per routing decision (Wall time (CLI))",
          "value": 2.6,
          "unit": "seconds",
          "display": "2.60 s",
          "n": 82,
          "range": [
            2.6,
            4.3
          ],
          "spanKind": "p50-p95",
          "context": "Claude Sonnet 5.5 · effort low · typed routing decisions, thinking on vs off"
        },
        {
          "studySlug": "haiku-thinking-on-off",
          "chartId": "haiku-thinking-router-latency",
          "series": "Model time (API)",
          "point": "Claude Haiku 4.5 (thinking off) · Claude Code",
          "metric": "haiku-thinking-router-latency/Model time (API)",
          "label": "Haiku thinking study: time per routing decision (Model time (API))",
          "value": 3.79,
          "unit": "seconds",
          "display": "3.79 s",
          "n": 82,
          "range": [
            3.79,
            7.43
          ],
          "spanKind": "p50-p95",
          "context": "Claude Haiku 4.5 · thinking off · typed routing decisions, thinking on vs off"
        },
        {
          "studySlug": "haiku-thinking-on-off",
          "chartId": "haiku-thinking-router-latency",
          "series": "Model time (API)",
          "point": "Claude Haiku 4.5 (thinking on) · Claude Code",
          "metric": "haiku-thinking-router-latency/Model time (API)",
          "label": "Haiku thinking study: time per routing decision (Model time (API))",
          "value": 10.51,
          "unit": "seconds",
          "display": "10.5 s",
          "n": 82,
          "range": [
            10.51,
            32.13
          ],
          "spanKind": "p50-p95",
          "context": "Claude Haiku 4.5 · thinking on · typed routing decisions, thinking on vs off"
        },
        {
          "studySlug": "haiku-thinking-on-off",
          "chartId": "haiku-thinking-router-latency",
          "series": "Model time (API)",
          "point": "Claude Sonnet 5.5 (low) · Claude Code",
          "metric": "haiku-thinking-router-latency/Model time (API)",
          "label": "Haiku thinking study: time per routing decision (Model time (API))",
          "value": 1.6,
          "unit": "seconds",
          "display": "1.60 s",
          "n": 82,
          "range": [
            1.6,
            2.58
          ],
          "spanKind": "p50-p95",
          "context": "Claude Sonnet 5.5 · effort low · typed routing decisions, thinking on vs off"
        },
        {
          "studySlug": "haiku-thinking-on-off",
          "chartId": "haiku-thinking-router-tokens",
          "series": "Thinking tokens",
          "point": "Claude Haiku 4.5 (thinking off) · Claude Code",
          "metric": "haiku-thinking-router-tokens/Thinking tokens",
          "label": "Haiku thinking study: thinking and visible output tokens per routing decision (Thinking tokens)",
          "value": 0,
          "unit": "tokens",
          "display": "0",
          "n": 82,
          "context": "Claude Haiku 4.5 · thinking off · typed routing decisions, thinking on vs off"
        },
        {
          "studySlug": "haiku-thinking-on-off",
          "chartId": "haiku-thinking-router-tokens",
          "series": "Thinking tokens",
          "point": "Claude Haiku 4.5 (thinking on) · Claude Code",
          "metric": "haiku-thinking-router-tokens/Thinking tokens",
          "label": "Haiku thinking study: thinking and visible output tokens per routing decision (Thinking tokens)",
          "value": 1101,
          "unit": "tokens",
          "display": "1,101",
          "n": 82,
          "context": "Claude Haiku 4.5 · thinking on · typed routing decisions, thinking on vs off"
        },
        {
          "studySlug": "haiku-thinking-on-off",
          "chartId": "haiku-thinking-router-tokens",
          "series": "Thinking tokens",
          "point": "Claude Sonnet 5.5 (low) · Claude Code",
          "metric": "haiku-thinking-router-tokens/Thinking tokens",
          "label": "Haiku thinking study: thinking and visible output tokens per routing decision (Thinking tokens)",
          "value": 2,
          "unit": "tokens",
          "display": "2",
          "n": 82,
          "context": "Claude Sonnet 5.5 · effort low · typed routing decisions, thinking on vs off"
        },
        {
          "studySlug": "haiku-thinking-on-off",
          "chartId": "haiku-thinking-router-tokens",
          "series": "Visible output tokens",
          "point": "Claude Haiku 4.5 (thinking off) · Claude Code",
          "metric": "haiku-thinking-router-tokens/Visible output tokens",
          "label": "Haiku thinking study: thinking and visible output tokens per routing decision (Visible output tokens)",
          "value": 366,
          "unit": "tokens",
          "display": "366",
          "n": 82,
          "context": "Claude Haiku 4.5 · thinking off · typed routing decisions, thinking on vs off"
        },
        {
          "studySlug": "haiku-thinking-on-off",
          "chartId": "haiku-thinking-router-tokens",
          "series": "Visible output tokens",
          "point": "Claude Haiku 4.5 (thinking on) · Claude Code",
          "metric": "haiku-thinking-router-tokens/Visible output tokens",
          "label": "Haiku thinking study: thinking and visible output tokens per routing decision (Visible output tokens)",
          "value": 318,
          "unit": "tokens",
          "display": "318",
          "n": 82,
          "context": "Claude Haiku 4.5 · thinking on · typed routing decisions, thinking on vs off"
        },
        {
          "studySlug": "haiku-thinking-on-off",
          "chartId": "haiku-thinking-router-tokens",
          "series": "Visible output tokens",
          "point": "Claude Sonnet 5.5 (low) · Claude Code",
          "metric": "haiku-thinking-router-tokens/Visible output tokens",
          "label": "Haiku thinking study: thinking and visible output tokens per routing decision (Visible output tokens)",
          "value": 105,
          "unit": "tokens",
          "display": "105",
          "n": 82,
          "context": "Claude Sonnet 5.5 · effort low · typed routing decisions, thinking on vs off"
        },
        {
          "studySlug": "haiku-thinking-on-off",
          "chartId": "haiku-thinking-router-cost",
          "series": "Cost per 1,000 decisions",
          "point": "Claude Haiku 4.5 (thinking off) · Claude Code",
          "metric": "haiku-thinking-router-cost",
          "label": "Haiku thinking study: list-price cost per 1,000 routing decisions (calculation)",
          "value": 3.364,
          "unit": "usd",
          "display": "$3.36",
          "n": 82,
          "calculation": true,
          "context": "Claude Haiku 4.5 · thinking off · typed routing decisions, thinking on vs off"
        },
        {
          "studySlug": "haiku-thinking-on-off",
          "chartId": "haiku-thinking-router-cost",
          "series": "Cost per 1,000 decisions",
          "point": "Claude Haiku 4.5 (thinking on) · Claude Code",
          "metric": "haiku-thinking-router-cost",
          "label": "Haiku thinking study: list-price cost per 1,000 routing decisions (calculation)",
          "value": 8.924,
          "unit": "usd",
          "display": "$8.92",
          "n": 82,
          "calculation": true,
          "context": "Claude Haiku 4.5 · thinking on · typed routing decisions, thinking on vs off"
        },
        {
          "studySlug": "haiku-thinking-on-off",
          "chartId": "haiku-thinking-router-cost",
          "series": "Cost per 1,000 decisions",
          "point": "Claude Sonnet 5.5 (low) · Claude Code",
          "metric": "haiku-thinking-router-cost",
          "label": "Haiku thinking study: list-price cost per 1,000 routing decisions (calculation)",
          "value": 7.324,
          "unit": "usd",
          "display": "$7.32",
          "n": 82,
          "calculation": true,
          "context": "Claude Sonnet 5.5 · effort low · typed routing decisions, thinking on vs off"
        },
        {
          "studySlug": "haiku-thinking-on-off",
          "chartId": "haiku-thinking-hard-pass",
          "series": "Strict pass",
          "point": "Claude Haiku 4.5 (thinking off) · Claude Code",
          "metric": "haiku-thinking-hard-pass/Strict pass",
          "label": "Haiku thinking study: pass rate on eight hard tasks (Strict pass)",
          "value": 0.1667,
          "unit": "rate",
          "display": "17% (4/24)",
          "n": 24,
          "ci": [
            0.0668,
            0.3585
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Haiku 4.5 · thinking off · eight hard validated tasks, thinking on vs off"
        },
        {
          "studySlug": "haiku-thinking-on-off",
          "chartId": "haiku-thinking-hard-pass",
          "series": "Strict pass",
          "point": "Claude Haiku 4.5 (thinking on) · Claude Code",
          "metric": "haiku-thinking-hard-pass/Strict pass",
          "label": "Haiku thinking study: pass rate on eight hard tasks (Strict pass)",
          "value": 0.4583,
          "unit": "rate",
          "display": "46% (11/24)",
          "n": 24,
          "ci": [
            0.2789,
            0.6493
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Haiku 4.5 · thinking on · eight hard validated tasks, thinking on vs off"
        },
        {
          "studySlug": "haiku-thinking-on-off",
          "chartId": "haiku-thinking-hard-pass",
          "series": "Lenient (format misses counted)",
          "point": "Claude Haiku 4.5 (thinking off) · Claude Code",
          "metric": "haiku-thinking-hard-pass/Lenient (format misses counted)",
          "label": "Haiku thinking study: pass rate on eight hard tasks (Lenient (format misses counted))",
          "value": 0.1667,
          "unit": "rate",
          "display": "17% (4/24)",
          "n": 24,
          "ci": [
            0.0668,
            0.3585
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Haiku 4.5 · thinking off · eight hard validated tasks, thinking on vs off"
        },
        {
          "studySlug": "haiku-thinking-on-off",
          "chartId": "haiku-thinking-hard-pass",
          "series": "Lenient (format misses counted)",
          "point": "Claude Haiku 4.5 (thinking on) · Claude Code",
          "metric": "haiku-thinking-hard-pass/Lenient (format misses counted)",
          "label": "Haiku thinking study: pass rate on eight hard tasks (Lenient (format misses counted))",
          "value": 0.6667,
          "unit": "rate",
          "display": "67% (16/24)",
          "n": 24,
          "ci": [
            0.4671,
            0.8203
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Haiku 4.5 · thinking on · eight hard validated tasks, thinking on vs off"
        },
        {
          "studySlug": "haiku-thinking-on-off",
          "chartId": "haiku-thinking-hard-time",
          "series": "Total time per call",
          "point": "Claude Haiku 4.5 (thinking off) · Claude Code",
          "metric": "haiku-thinking-hard-time",
          "label": "Haiku thinking study: total time per call on hard tasks",
          "value": 2.95,
          "unit": "seconds",
          "display": "2.95 s",
          "n": 24,
          "range": [
            1.7,
            13.01
          ],
          "spanKind": "minmax",
          "context": "Claude Haiku 4.5 · thinking off · eight hard validated tasks, thinking on vs off"
        },
        {
          "studySlug": "haiku-thinking-on-off",
          "chartId": "haiku-thinking-hard-time",
          "series": "Total time per call",
          "point": "Claude Haiku 4.5 (thinking on) · Claude Code",
          "metric": "haiku-thinking-hard-time",
          "label": "Haiku thinking study: total time per call on hard tasks",
          "value": 39.01,
          "unit": "seconds",
          "display": "39.0 s",
          "n": 24,
          "range": [
            15.27,
            75.13
          ],
          "spanKind": "minmax",
          "context": "Claude Haiku 4.5 · thinking on · eight hard validated tasks, thinking on vs off"
        },
        {
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-pass-rate",
          "series": "Strict pass: the whole reply is the right JSON",
          "point": "Claude Haiku 4.5 (instructions) · Claude Code",
          "metric": "structured-output-pass-rate/Strict pass: the whole reply is the right JSON",
          "label": "Does a JSON schema raise the pass rate? Instructions vs schema mode (Strict pass: the whole reply is the right JSON)",
          "value": 0,
          "unit": "rate",
          "display": "0% (0/24)",
          "n": 24,
          "ci": [
            0,
            0.138
          ],
          "spanKind": "ci95",
          "calculation": true,
          "polarity": "higher",
          "context": "Claude Haiku 4.5 · instructions"
        },
        {
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-pass-rate",
          "series": "Strict pass: the whole reply is the right JSON",
          "point": "Claude Haiku 4.5 (JSON schema) · Claude Code",
          "metric": "structured-output-pass-rate/Strict pass: the whole reply is the right JSON",
          "label": "Does a JSON schema raise the pass rate? Instructions vs schema mode (Strict pass: the whole reply is the right JSON)",
          "value": 0.75,
          "unit": "rate",
          "display": "75% (18/24)",
          "n": 24,
          "ci": [
            0.551,
            0.88
          ],
          "spanKind": "ci95",
          "calculation": true,
          "polarity": "higher",
          "context": "Claude Haiku 4.5 · JSON schema"
        },
        {
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-pass-rate",
          "series": "Strict pass: the whole reply is the right JSON",
          "point": "Claude Sonnet 5.5 (instructions) · Claude Code",
          "metric": "structured-output-pass-rate/Strict pass: the whole reply is the right JSON",
          "label": "Does a JSON schema raise the pass rate? Instructions vs schema mode (Strict pass: the whole reply is the right JSON)",
          "value": 1,
          "unit": "rate",
          "display": "100% (12/12)",
          "n": 12,
          "ci": [
            0.7575,
            1
          ],
          "spanKind": "ci95",
          "calculation": true,
          "polarity": "higher",
          "context": "Claude Sonnet 5.5 · instructions"
        },
        {
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-pass-rate",
          "series": "Strict pass: the whole reply is the right JSON",
          "point": "Claude Sonnet 5.5 (JSON schema) · Claude Code",
          "metric": "structured-output-pass-rate/Strict pass: the whole reply is the right JSON",
          "label": "Does a JSON schema raise the pass rate? Instructions vs schema mode (Strict pass: the whole reply is the right JSON)",
          "value": 1,
          "unit": "rate",
          "display": "100% (12/12)",
          "n": 12,
          "ci": [
            0.7575,
            1
          ],
          "spanKind": "ci95",
          "calculation": true,
          "polarity": "higher",
          "context": "Claude Sonnet 5.5 · JSON schema"
        },
        {
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-pass-rate",
          "series": "Right answer in any format (strict pass or format miss)",
          "point": "Claude Haiku 4.5 (instructions) · Claude Code",
          "metric": "structured-output-pass-rate/Right answer in any format (strict pass or format miss)",
          "label": "Does a JSON schema raise the pass rate? Instructions vs schema mode (Right answer in any format (strict pass or format miss))",
          "value": 0.7083,
          "unit": "rate",
          "display": "71% (17/24)",
          "n": 24,
          "ci": [
            0.5083,
            0.8509
          ],
          "spanKind": "ci95",
          "calculation": true,
          "polarity": "higher",
          "context": "Claude Haiku 4.5 · instructions"
        },
        {
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-pass-rate",
          "series": "Right answer in any format (strict pass or format miss)",
          "point": "Claude Haiku 4.5 (JSON schema) · Claude Code",
          "metric": "structured-output-pass-rate/Right answer in any format (strict pass or format miss)",
          "label": "Does a JSON schema raise the pass rate? Instructions vs schema mode (Right answer in any format (strict pass or format miss))",
          "value": 0.75,
          "unit": "rate",
          "display": "75% (18/24)",
          "n": 24,
          "ci": [
            0.551,
            0.88
          ],
          "spanKind": "ci95",
          "calculation": true,
          "polarity": "higher",
          "context": "Claude Haiku 4.5 · JSON schema"
        },
        {
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-pass-rate",
          "series": "Right answer in any format (strict pass or format miss)",
          "point": "Claude Sonnet 5.5 (instructions) · Claude Code",
          "metric": "structured-output-pass-rate/Right answer in any format (strict pass or format miss)",
          "label": "Does a JSON schema raise the pass rate? Instructions vs schema mode (Right answer in any format (strict pass or format miss))",
          "value": 1,
          "unit": "rate",
          "display": "100% (12/12)",
          "n": 12,
          "ci": [
            0.7575,
            1
          ],
          "spanKind": "ci95",
          "calculation": true,
          "polarity": "higher",
          "context": "Claude Sonnet 5.5 · instructions"
        },
        {
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-pass-rate",
          "series": "Right answer in any format (strict pass or format miss)",
          "point": "Claude Sonnet 5.5 (JSON schema) · Claude Code",
          "metric": "structured-output-pass-rate/Right answer in any format (strict pass or format miss)",
          "label": "Does a JSON schema raise the pass rate? Instructions vs schema mode (Right answer in any format (strict pass or format miss))",
          "value": 1,
          "unit": "rate",
          "display": "100% (12/12)",
          "n": 12,
          "ci": [
            0.7575,
            1
          ],
          "spanKind": "ci95",
          "calculation": true,
          "polarity": "higher",
          "context": "Claude Sonnet 5.5 · JSON schema"
        },
        {
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-outcomes",
          "series": "Strict pass",
          "point": "Claude Haiku 4.5 (instructions) · Claude Code",
          "metric": "structured-output-outcomes/Strict pass",
          "label": "What each call produced: strict pass, format miss, wrong values or error (Strict pass)",
          "value": 0,
          "unit": "count",
          "display": "0",
          "n": 24,
          "context": "Claude Haiku 4.5 · instructions"
        },
        {
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-outcomes",
          "series": "Strict pass",
          "point": "Claude Haiku 4.5 (JSON schema) · Claude Code",
          "metric": "structured-output-outcomes/Strict pass",
          "label": "What each call produced: strict pass, format miss, wrong values or error (Strict pass)",
          "value": 18,
          "unit": "count",
          "display": "18",
          "n": 24,
          "context": "Claude Haiku 4.5 · JSON schema"
        },
        {
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-outcomes",
          "series": "Strict pass",
          "point": "Claude Sonnet 5.5 (instructions) · Claude Code",
          "metric": "structured-output-outcomes/Strict pass",
          "label": "What each call produced: strict pass, format miss, wrong values or error (Strict pass)",
          "value": 12,
          "unit": "count",
          "display": "12",
          "n": 12,
          "context": "Claude Sonnet 5.5 · instructions"
        },
        {
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-outcomes",
          "series": "Strict pass",
          "point": "Claude Sonnet 5.5 (JSON schema) · Claude Code",
          "metric": "structured-output-outcomes/Strict pass",
          "label": "What each call produced: strict pass, format miss, wrong values or error (Strict pass)",
          "value": 12,
          "unit": "count",
          "display": "12",
          "n": 12,
          "context": "Claude Sonnet 5.5 · JSON schema"
        },
        {
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-outcomes",
          "series": "Format miss",
          "point": "Claude Haiku 4.5 (instructions) · Claude Code",
          "metric": "structured-output-outcomes/Format miss",
          "label": "What each call produced: strict pass, format miss, wrong values or error (Format miss)",
          "value": 17,
          "unit": "count",
          "display": "17",
          "n": 24,
          "context": "Claude Haiku 4.5 · instructions"
        },
        {
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-outcomes",
          "series": "Format miss",
          "point": "Claude Haiku 4.5 (JSON schema) · Claude Code",
          "metric": "structured-output-outcomes/Format miss",
          "label": "What each call produced: strict pass, format miss, wrong values or error (Format miss)",
          "value": 0,
          "unit": "count",
          "display": "0",
          "n": 24,
          "context": "Claude Haiku 4.5 · JSON schema"
        },
        {
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-outcomes",
          "series": "Format miss",
          "point": "Claude Sonnet 5.5 (instructions) · Claude Code",
          "metric": "structured-output-outcomes/Format miss",
          "label": "What each call produced: strict pass, format miss, wrong values or error (Format miss)",
          "value": 0,
          "unit": "count",
          "display": "0",
          "n": 12,
          "context": "Claude Sonnet 5.5 · instructions"
        },
        {
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-outcomes",
          "series": "Format miss",
          "point": "Claude Sonnet 5.5 (JSON schema) · Claude Code",
          "metric": "structured-output-outcomes/Format miss",
          "label": "What each call produced: strict pass, format miss, wrong values or error (Format miss)",
          "value": 0,
          "unit": "count",
          "display": "0",
          "n": 12,
          "context": "Claude Sonnet 5.5 · JSON schema"
        },
        {
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-outcomes",
          "series": "Wrong values",
          "point": "Claude Haiku 4.5 (instructions) · Claude Code",
          "metric": "structured-output-outcomes/Wrong values",
          "label": "What each call produced: strict pass, format miss, wrong values or error (Wrong values)",
          "value": 7,
          "unit": "count",
          "display": "7",
          "n": 24,
          "context": "Claude Haiku 4.5 · instructions"
        },
        {
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-outcomes",
          "series": "Wrong values",
          "point": "Claude Haiku 4.5 (JSON schema) · Claude Code",
          "metric": "structured-output-outcomes/Wrong values",
          "label": "What each call produced: strict pass, format miss, wrong values or error (Wrong values)",
          "value": 6,
          "unit": "count",
          "display": "6",
          "n": 24,
          "context": "Claude Haiku 4.5 · JSON schema"
        },
        {
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-outcomes",
          "series": "Wrong values",
          "point": "Claude Sonnet 5.5 (instructions) · Claude Code",
          "metric": "structured-output-outcomes/Wrong values",
          "label": "What each call produced: strict pass, format miss, wrong values or error (Wrong values)",
          "value": 0,
          "unit": "count",
          "display": "0",
          "n": 12,
          "context": "Claude Sonnet 5.5 · instructions"
        },
        {
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-outcomes",
          "series": "Wrong values",
          "point": "Claude Sonnet 5.5 (JSON schema) · Claude Code",
          "metric": "structured-output-outcomes/Wrong values",
          "label": "What each call produced: strict pass, format miss, wrong values or error (Wrong values)",
          "value": 0,
          "unit": "count",
          "display": "0",
          "n": 12,
          "context": "Claude Sonnet 5.5 · JSON schema"
        },
        {
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-outcomes",
          "series": "Error",
          "point": "Claude Haiku 4.5 (instructions) · Claude Code",
          "metric": "structured-output-outcomes/Error",
          "label": "What each call produced: strict pass, format miss, wrong values or error (Error)",
          "value": 0,
          "unit": "count",
          "display": "0",
          "n": 24,
          "context": "Claude Haiku 4.5 · instructions"
        },
        {
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-outcomes",
          "series": "Error",
          "point": "Claude Haiku 4.5 (JSON schema) · Claude Code",
          "metric": "structured-output-outcomes/Error",
          "label": "What each call produced: strict pass, format miss, wrong values or error (Error)",
          "value": 0,
          "unit": "count",
          "display": "0",
          "n": 24,
          "context": "Claude Haiku 4.5 · JSON schema"
        },
        {
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-outcomes",
          "series": "Error",
          "point": "Claude Sonnet 5.5 (instructions) · Claude Code",
          "metric": "structured-output-outcomes/Error",
          "label": "What each call produced: strict pass, format miss, wrong values or error (Error)",
          "value": 0,
          "unit": "count",
          "display": "0",
          "n": 12,
          "context": "Claude Sonnet 5.5 · instructions"
        },
        {
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-outcomes",
          "series": "Error",
          "point": "Claude Sonnet 5.5 (JSON schema) · Claude Code",
          "metric": "structured-output-outcomes/Error",
          "label": "What each call produced: strict pass, format miss, wrong values or error (Error)",
          "value": 0,
          "unit": "count",
          "display": "0",
          "n": 12,
          "context": "Claude Sonnet 5.5 · JSON schema"
        },
        {
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-time",
          "series": "Median time per call (the three prompts pooled)",
          "point": "Claude Haiku 4.5 (instructions) · Claude Code",
          "metric": "structured-output-time",
          "label": "Time per call, instructions vs schema mode",
          "value": 9.52,
          "unit": "seconds",
          "display": "9.52 s",
          "n": 24,
          "range": [
            5.67,
            17
          ],
          "spanKind": "minmax",
          "context": "Claude Haiku 4.5 · instructions"
        },
        {
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-time",
          "series": "Median time per call (the three prompts pooled)",
          "point": "Claude Haiku 4.5 (JSON schema) · Claude Code",
          "metric": "structured-output-time",
          "label": "Time per call, instructions vs schema mode",
          "value": 8.46,
          "unit": "seconds",
          "display": "8.46 s",
          "n": 24,
          "range": [
            5.9,
            12.23
          ],
          "spanKind": "minmax",
          "context": "Claude Haiku 4.5 · JSON schema"
        },
        {
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-time",
          "series": "Median time per call (the three prompts pooled)",
          "point": "Claude Sonnet 5.5 (instructions) · Claude Code",
          "metric": "structured-output-time",
          "label": "Time per call, instructions vs schema mode",
          "value": 3.52,
          "unit": "seconds",
          "display": "3.52 s",
          "n": 12,
          "range": [
            2.67,
            4.12
          ],
          "spanKind": "minmax",
          "context": "Claude Sonnet 5.5 · instructions"
        },
        {
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-time",
          "series": "Median time per call (the three prompts pooled)",
          "point": "Claude Sonnet 5.5 (JSON schema) · Claude Code",
          "metric": "structured-output-time",
          "label": "Time per call, instructions vs schema mode",
          "value": 4.2,
          "unit": "seconds",
          "display": "4.20 s",
          "n": 12,
          "range": [
            2.95,
            6.14
          ],
          "spanKind": "minmax",
          "context": "Claude Sonnet 5.5 · JSON schema"
        },
        {
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-tokens",
          "series": "Median output tokens per call",
          "point": "Claude Haiku 4.5 (instructions) · Claude Code",
          "metric": "structured-output-tokens/Median output tokens per call",
          "label": "Output and reasoning tokens per call, instructions vs schema mode (Median output tokens per call)",
          "value": 1128,
          "unit": "tokens",
          "display": "1,128",
          "n": 24,
          "context": "Claude Haiku 4.5 · instructions"
        },
        {
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-tokens",
          "series": "Median output tokens per call",
          "point": "Claude Haiku 4.5 (JSON schema) · Claude Code",
          "metric": "structured-output-tokens/Median output tokens per call",
          "label": "Output and reasoning tokens per call, instructions vs schema mode (Median output tokens per call)",
          "value": 1036,
          "unit": "tokens",
          "display": "1,036",
          "n": 24,
          "context": "Claude Haiku 4.5 · JSON schema"
        },
        {
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-tokens",
          "series": "Median output tokens per call",
          "point": "Claude Sonnet 5.5 (instructions) · Claude Code",
          "metric": "structured-output-tokens/Median output tokens per call",
          "label": "Output and reasoning tokens per call, instructions vs schema mode (Median output tokens per call)",
          "value": 368,
          "unit": "tokens",
          "display": "368",
          "n": 12,
          "context": "Claude Sonnet 5.5 · instructions"
        },
        {
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-tokens",
          "series": "Median output tokens per call",
          "point": "Claude Sonnet 5.5 (JSON schema) · Claude Code",
          "metric": "structured-output-tokens/Median output tokens per call",
          "label": "Output and reasoning tokens per call, instructions vs schema mode (Median output tokens per call)",
          "value": 424,
          "unit": "tokens",
          "display": "424",
          "n": 12,
          "context": "Claude Sonnet 5.5 · JSON schema"
        },
        {
          "studySlug": "routing-holdout",
          "chartId": "routing-holdout-exact",
          "series": "Exact rate",
          "point": "Claude Haiku 4.5 · Claude Code",
          "metric": "routing-holdout-exact",
          "label": "Unseen routing decisions answered exactly right",
          "value": 0.7857,
          "unit": "rate",
          "display": "79% (44/56)",
          "n": 56,
          "ci": [
            0.6618,
            0.8729
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Haiku 4.5"
        },
        {
          "studySlug": "routing-holdout",
          "chartId": "routing-holdout-exact",
          "series": "Exact rate",
          "point": "Claude Sonnet 5.5 (low) · Claude Code",
          "metric": "routing-holdout-exact",
          "label": "Unseen routing decisions answered exactly right",
          "value": 0.875,
          "unit": "rate",
          "display": "88% (49/56)",
          "n": 56,
          "ci": [
            0.7637,
            0.9381
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Sonnet 5.5 · effort low"
        },
        {
          "studySlug": "routing-holdout",
          "chartId": "routing-holdout-key-accuracy",
          "series": "Key accuracy",
          "point": "Claude Haiku 4.5 · Claude Code",
          "metric": "routing-holdout-key-accuracy",
          "label": "Per-question accuracy on unseen decisions",
          "value": 0.816,
          "unit": "rate",
          "display": "82% (102/125)",
          "n": 125,
          "ci": [
            0.739,
            0.8741
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Haiku 4.5"
        },
        {
          "studySlug": "routing-holdout",
          "chartId": "routing-holdout-key-accuracy",
          "series": "Key accuracy",
          "point": "Claude Sonnet 5.5 (low) · Claude Code",
          "metric": "routing-holdout-key-accuracy",
          "label": "Per-question accuracy on unseen decisions",
          "value": 0.92,
          "unit": "rate",
          "display": "92% (115/125)",
          "n": 125,
          "ci": [
            0.859,
            0.956
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Sonnet 5.5 · effort low"
        },
        {
          "studySlug": "routing-holdout",
          "chartId": "routing-holdout-by-purpose",
          "series": "Claude Haiku 4.5 · Claude Code",
          "point": "Failure class",
          "metric": "routing-holdout-by-purpose/Failure class",
          "label": "Exact rate on unseen decisions, by decision type: Failure class",
          "value": 0.9286,
          "unit": "rate",
          "display": "93% (13/14)",
          "n": 14,
          "ci": [
            0.6853,
            0.9873
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Haiku 4.5"
        },
        {
          "studySlug": "routing-holdout",
          "chartId": "routing-holdout-by-purpose",
          "series": "Claude Haiku 4.5 · Claude Code",
          "point": "Message intent",
          "metric": "routing-holdout-by-purpose/Message intent",
          "label": "Exact rate on unseen decisions, by decision type: Message intent",
          "value": 0.9286,
          "unit": "rate",
          "display": "93% (13/14)",
          "n": 14,
          "ci": [
            0.6853,
            0.9873
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Haiku 4.5"
        },
        {
          "studySlug": "routing-holdout",
          "chartId": "routing-holdout-by-purpose",
          "series": "Claude Haiku 4.5 · Claude Code",
          "point": "Is it a rule?",
          "metric": "routing-holdout-by-purpose/Is it a rule?",
          "label": "Exact rate on unseen decisions, by decision type: Is it a rule?",
          "value": 0.9286,
          "unit": "rate",
          "display": "93% (13/14)",
          "n": 14,
          "ci": [
            0.6853,
            0.9873
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Haiku 4.5"
        },
        {
          "studySlug": "routing-holdout",
          "chartId": "routing-holdout-by-purpose",
          "series": "Claude Haiku 4.5 · Claude Code",
          "point": "Context shape",
          "metric": "routing-holdout-by-purpose/Context shape",
          "label": "Exact rate on unseen decisions, by decision type: Context shape",
          "value": 0.3571,
          "unit": "rate",
          "display": "36% (5/14)",
          "n": 14,
          "ci": [
            0.1634,
            0.6124
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Haiku 4.5"
        },
        {
          "studySlug": "routing-holdout",
          "chartId": "routing-holdout-by-purpose",
          "series": "Claude Sonnet 5.5 (low) · Claude Code",
          "point": "Failure class",
          "metric": "routing-holdout-by-purpose/Failure class",
          "label": "Exact rate on unseen decisions, by decision type: Failure class",
          "value": 1,
          "unit": "rate",
          "display": "100% (14/14)",
          "n": 14,
          "ci": [
            0.7847,
            1
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Sonnet 5.5 · effort low"
        },
        {
          "studySlug": "routing-holdout",
          "chartId": "routing-holdout-by-purpose",
          "series": "Claude Sonnet 5.5 (low) · Claude Code",
          "point": "Message intent",
          "metric": "routing-holdout-by-purpose/Message intent",
          "label": "Exact rate on unseen decisions, by decision type: Message intent",
          "value": 1,
          "unit": "rate",
          "display": "100% (14/14)",
          "n": 14,
          "ci": [
            0.7847,
            1
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Sonnet 5.5 · effort low"
        },
        {
          "studySlug": "routing-holdout",
          "chartId": "routing-holdout-by-purpose",
          "series": "Claude Sonnet 5.5 (low) · Claude Code",
          "point": "Is it a rule?",
          "metric": "routing-holdout-by-purpose/Is it a rule?",
          "label": "Exact rate on unseen decisions, by decision type: Is it a rule?",
          "value": 0.9286,
          "unit": "rate",
          "display": "93% (13/14)",
          "n": 14,
          "ci": [
            0.6853,
            0.9873
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Sonnet 5.5 · effort low"
        },
        {
          "studySlug": "routing-holdout",
          "chartId": "routing-holdout-by-purpose",
          "series": "Claude Sonnet 5.5 (low) · Claude Code",
          "point": "Context shape",
          "metric": "routing-holdout-by-purpose/Context shape",
          "label": "Exact rate on unseen decisions, by decision type: Context shape",
          "value": 0.5714,
          "unit": "rate",
          "display": "57% (8/14)",
          "n": 14,
          "ci": [
            0.3259,
            0.7862
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Sonnet 5.5 · effort low"
        },
        {
          "studySlug": "routing-holdout",
          "chartId": "routing-holdout-tuned-vs-unseen",
          "series": "Tuned set (routing-jev-vs-llm)",
          "point": "Claude Haiku 4.5 · Claude Code",
          "metric": "routing-holdout-tuned-vs-unseen/Tuned set (routing-jev-vs-llm)",
          "label": "Tuned case set vs unseen holdout: exact rate per router (Tuned set (routing-jev-vs-llm))",
          "value": 0.8902,
          "unit": "rate",
          "display": "89% (73/82)",
          "n": 82,
          "ci": [
            0.8044,
            0.9412
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Haiku 4.5"
        },
        {
          "studySlug": "routing-holdout",
          "chartId": "routing-holdout-tuned-vs-unseen",
          "series": "Tuned set (routing-jev-vs-llm)",
          "point": "Claude Sonnet 5.5 (low) · Claude Code",
          "metric": "routing-holdout-tuned-vs-unseen/Tuned set (routing-jev-vs-llm)",
          "label": "Tuned case set vs unseen holdout: exact rate per router (Tuned set (routing-jev-vs-llm))",
          "value": 0.939,
          "unit": "rate",
          "display": "94% (77/82)",
          "n": 82,
          "ci": [
            0.8651,
            0.9737
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Sonnet 5.5 · effort low"
        },
        {
          "studySlug": "routing-holdout",
          "chartId": "routing-holdout-tuned-vs-unseen",
          "series": "Unseen holdout",
          "point": "Claude Haiku 4.5 · Claude Code",
          "metric": "routing-holdout-tuned-vs-unseen/Unseen holdout",
          "label": "Tuned case set vs unseen holdout: exact rate per router (Unseen holdout)",
          "value": 0.7857,
          "unit": "rate",
          "display": "79% (44/56)",
          "n": 56,
          "ci": [
            0.6618,
            0.8729
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Haiku 4.5"
        },
        {
          "studySlug": "routing-holdout",
          "chartId": "routing-holdout-tuned-vs-unseen",
          "series": "Unseen holdout",
          "point": "Claude Sonnet 5.5 (low) · Claude Code",
          "metric": "routing-holdout-tuned-vs-unseen/Unseen holdout",
          "label": "Tuned case set vs unseen holdout: exact rate per router (Unseen holdout)",
          "value": 0.875,
          "unit": "rate",
          "display": "88% (49/56)",
          "n": 56,
          "ci": [
            0.7637,
            0.9381
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Sonnet 5.5 · effort low"
        },
        {
          "studySlug": "routing-holdout",
          "chartId": "routing-holdout-latency",
          "series": "Wall time",
          "point": "Claude Haiku 4.5 · Claude Code",
          "metric": "routing-holdout-latency/Wall time",
          "label": "Time per routing decision, by route (Wall time)",
          "value": 9.444,
          "unit": "seconds",
          "display": "9.44 s",
          "n": 56,
          "range": [
            9.444,
            25.413
          ],
          "spanKind": "p50-p95",
          "context": "Claude Haiku 4.5"
        },
        {
          "studySlug": "routing-holdout",
          "chartId": "routing-holdout-latency",
          "series": "Wall time",
          "point": "Claude Sonnet 5.5 (low) · Claude Code",
          "metric": "routing-holdout-latency/Wall time",
          "label": "Time per routing decision, by route (Wall time)",
          "value": 2.359,
          "unit": "seconds",
          "display": "2.36 s",
          "n": 56,
          "range": [
            2.359,
            3.657
          ],
          "spanKind": "p50-p95",
          "context": "Claude Sonnet 5.5 · effort low"
        },
        {
          "studySlug": "routing-holdout",
          "chartId": "routing-holdout-latency",
          "series": "Model time (API, CLI-reported)",
          "point": "Claude Haiku 4.5 · Claude Code",
          "metric": "routing-holdout-latency/Model time (API, CLI-reported)",
          "label": "Time per routing decision, by route (Model time (API, CLI-reported))",
          "value": 7.522,
          "unit": "seconds",
          "display": "7.52 s",
          "n": 56,
          "range": [
            7.522,
            23.913
          ],
          "spanKind": "p50-p95",
          "context": "Claude Haiku 4.5"
        },
        {
          "studySlug": "routing-holdout",
          "chartId": "routing-holdout-latency",
          "series": "Model time (API, CLI-reported)",
          "point": "Claude Sonnet 5.5 (low) · Claude Code",
          "metric": "routing-holdout-latency/Model time (API, CLI-reported)",
          "label": "Time per routing decision, by route (Model time (API, CLI-reported))",
          "value": 1.485,
          "unit": "seconds",
          "display": "1.49 s",
          "n": 56,
          "range": [
            1.485,
            2.377
          ],
          "spanKind": "p50-p95",
          "context": "Claude Sonnet 5.5 · effort low"
        },
        {
          "studySlug": "routing-holdout",
          "chartId": "routing-holdout-cost-per-1000",
          "series": "Cost",
          "point": "Claude Haiku 4.5 · Claude Code",
          "metric": "routing-holdout-cost-per-1000",
          "label": "Cost per 1,000 unseen routing decisions",
          "value": 7.129,
          "unit": "usd",
          "display": "$7.13",
          "n": 56,
          "calculation": true,
          "context": "Claude Haiku 4.5"
        },
        {
          "studySlug": "routing-holdout",
          "chartId": "routing-holdout-cost-per-1000",
          "series": "Cost",
          "point": "Claude Sonnet 5.5 (low) · Claude Code",
          "metric": "routing-holdout-cost-per-1000",
          "label": "Cost per 1,000 unseen routing decisions",
          "value": 7.244,
          "unit": "usd",
          "display": "$7.24",
          "n": 56,
          "calculation": true,
          "context": "Claude Sonnet 5.5 · effort low"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-share",
          "series": "Median call",
          "point": "Claude Haiku 4.5 · Claude Code",
          "metric": "thinking-bill-share",
          "label": "Reasoning share of output tokens per call on hard tasks (calculation)",
          "value": 91.68,
          "unit": "percent",
          "display": "91.7%",
          "n": 24,
          "range": [
            76.46,
            99.27
          ],
          "spanKind": "minmax",
          "calculation": true,
          "polarity": "none",
          "context": "Claude Haiku 4.5"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-share",
          "series": "Median call",
          "point": "Claude Fable 5.1 · Claude Code",
          "metric": "thinking-bill-share",
          "label": "Reasoning share of output tokens per call on hard tasks (calculation)",
          "value": 64.24,
          "unit": "percent",
          "display": "64.2%",
          "n": 24,
          "range": [
            23.44,
            97.19
          ],
          "spanKind": "minmax",
          "calculation": true,
          "polarity": "none",
          "context": "Claude Fable 5.1"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-share",
          "series": "Median call",
          "point": "Claude Opus 5.5 · Claude Code",
          "metric": "thinking-bill-share",
          "label": "Reasoning share of output tokens per call on hard tasks (calculation)",
          "value": 54.79,
          "unit": "percent",
          "display": "54.8%",
          "n": 24,
          "range": [
            29.92,
            95.6
          ],
          "spanKind": "minmax",
          "calculation": true,
          "polarity": "none",
          "context": "Claude Opus 5.5"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-share",
          "series": "Median call",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "thinking-bill-share",
          "label": "Reasoning share of output tokens per call on hard tasks (calculation)",
          "value": 54.54,
          "unit": "percent",
          "display": "54.5%",
          "n": 24,
          "range": [
            0,
            95.91
          ],
          "spanKind": "minmax",
          "calculation": true,
          "polarity": "none",
          "context": "Claude Sonnet 5.5"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-share",
          "series": "Median call",
          "point": "Claude Opus 5.5 (high) · Claude Code",
          "metric": "thinking-bill-share",
          "label": "Reasoning share of output tokens per call on hard tasks (calculation)",
          "value": 54.43,
          "unit": "percent",
          "display": "54.4%",
          "n": 24,
          "range": [
            36.14,
            96.23
          ],
          "spanKind": "minmax",
          "calculation": true,
          "polarity": "none",
          "context": "Claude Opus 5.5 · effort high"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-cost-per-call",
          "series": "Reasoning (output tokens)",
          "point": "Claude Fable 5.1 · Claude Code",
          "metric": "thinking-bill-cost-per-call/Reasoning (output tokens)",
          "label": "List-price cost per call: reasoning, remaining output and input (calculation) (Reasoning (output tokens))",
          "value": 0.053696,
          "unit": "usd",
          "display": "$0.054",
          "n": 24,
          "calculation": true,
          "polarity": "none",
          "context": "Claude Fable 5.1"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-cost-per-call",
          "series": "Reasoning (output tokens)",
          "point": "Claude Opus 5.5 (high) · Claude Code",
          "metric": "thinking-bill-cost-per-call/Reasoning (output tokens)",
          "label": "List-price cost per call: reasoning, remaining output and input (calculation) (Reasoning (output tokens))",
          "value": 0.017969,
          "unit": "usd",
          "display": "$0.018",
          "n": 24,
          "calculation": true,
          "polarity": "none",
          "context": "Claude Opus 5.5 · effort high"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-cost-per-call",
          "series": "Reasoning (output tokens)",
          "point": "Claude Haiku 4.5 · Claude Code",
          "metric": "thinking-bill-cost-per-call/Reasoning (output tokens)",
          "label": "List-price cost per call: reasoning, remaining output and input (calculation) (Reasoning (output tokens))",
          "value": 0.024492,
          "unit": "usd",
          "display": "$0.024",
          "n": 24,
          "calculation": true,
          "polarity": "none",
          "context": "Claude Haiku 4.5"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-cost-per-call",
          "series": "Reasoning (output tokens)",
          "point": "Claude Opus 5.5 · Claude Code",
          "metric": "thinking-bill-cost-per-call/Reasoning (output tokens)",
          "label": "List-price cost per call: reasoning, remaining output and input (calculation) (Reasoning (output tokens))",
          "value": 0.012528,
          "unit": "usd",
          "display": "$0.013",
          "n": 24,
          "calculation": true,
          "polarity": "none",
          "context": "Claude Opus 5.5"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-cost-per-call",
          "series": "Reasoning (output tokens)",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "thinking-bill-cost-per-call/Reasoning (output tokens)",
          "label": "List-price cost per call: reasoning, remaining output and input (calculation) (Reasoning (output tokens))",
          "value": 0.006665,
          "unit": "usd",
          "display": "$0.0067",
          "n": 24,
          "calculation": true,
          "polarity": "none",
          "context": "Claude Sonnet 5.5"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-cost-per-call",
          "series": "Remaining output (visible-answer estimate)",
          "point": "Claude Fable 5.1 · Claude Code",
          "metric": "thinking-bill-cost-per-call/Remaining output (visible-answer estimate)",
          "label": "List-price cost per call: reasoning, remaining output and input (calculation) (Remaining output (visible-answer estimate))",
          "value": 0.018702,
          "unit": "usd",
          "display": "$0.019",
          "n": 24,
          "calculation": true,
          "polarity": "none",
          "context": "Claude Fable 5.1"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-cost-per-call",
          "series": "Remaining output (visible-answer estimate)",
          "point": "Claude Opus 5.5 (high) · Claude Code",
          "metric": "thinking-bill-cost-per-call/Remaining output (visible-answer estimate)",
          "label": "List-price cost per call: reasoning, remaining output and input (calculation) (Remaining output (visible-answer estimate))",
          "value": 0.00799,
          "unit": "usd",
          "display": "$0.0080",
          "n": 24,
          "calculation": true,
          "polarity": "none",
          "context": "Claude Opus 5.5 · effort high"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-cost-per-call",
          "series": "Remaining output (visible-answer estimate)",
          "point": "Claude Haiku 4.5 · Claude Code",
          "metric": "thinking-bill-cost-per-call/Remaining output (visible-answer estimate)",
          "label": "List-price cost per call: reasoning, remaining output and input (calculation) (Remaining output (visible-answer estimate))",
          "value": 0.001795,
          "unit": "usd",
          "display": "$0.0018",
          "n": 24,
          "calculation": true,
          "polarity": "none",
          "context": "Claude Haiku 4.5"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-cost-per-call",
          "series": "Remaining output (visible-answer estimate)",
          "point": "Claude Opus 5.5 · Claude Code",
          "metric": "thinking-bill-cost-per-call/Remaining output (visible-answer estimate)",
          "label": "List-price cost per call: reasoning, remaining output and input (calculation) (Remaining output (visible-answer estimate))",
          "value": 0.008003,
          "unit": "usd",
          "display": "$0.0080",
          "n": 24,
          "calculation": true,
          "polarity": "none",
          "context": "Claude Opus 5.5"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-cost-per-call",
          "series": "Remaining output (visible-answer estimate)",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "thinking-bill-cost-per-call/Remaining output (visible-answer estimate)",
          "label": "List-price cost per call: reasoning, remaining output and input (calculation) (Remaining output (visible-answer estimate))",
          "value": 0.003672,
          "unit": "usd",
          "display": "$0.0037",
          "n": 24,
          "calculation": true,
          "polarity": "none",
          "context": "Claude Sonnet 5.5"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-cost-per-call",
          "series": "Input (prompt, cache priced)",
          "point": "Claude Fable 5.1 · Claude Code",
          "metric": "thinking-bill-cost-per-call/Input (prompt, cache priced)",
          "label": "List-price cost per call: reasoning, remaining output and input (calculation) (Input (prompt, cache priced))",
          "value": 0.02091,
          "unit": "usd",
          "display": "$0.021",
          "n": 24,
          "calculation": true,
          "polarity": "none",
          "context": "Claude Fable 5.1"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-cost-per-call",
          "series": "Input (prompt, cache priced)",
          "point": "Claude Opus 5.5 (high) · Claude Code",
          "metric": "thinking-bill-cost-per-call/Input (prompt, cache priced)",
          "label": "List-price cost per call: reasoning, remaining output and input (calculation) (Input (prompt, cache priced))",
          "value": 0.007407,
          "unit": "usd",
          "display": "$0.0074",
          "n": 24,
          "calculation": true,
          "polarity": "none",
          "context": "Claude Opus 5.5 · effort high"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-cost-per-call",
          "series": "Input (prompt, cache priced)",
          "point": "Claude Haiku 4.5 · Claude Code",
          "metric": "thinking-bill-cost-per-call/Input (prompt, cache priced)",
          "label": "List-price cost per call: reasoning, remaining output and input (calculation) (Input (prompt, cache priced))",
          "value": 0.00451,
          "unit": "usd",
          "display": "$0.0045",
          "n": 24,
          "calculation": true,
          "polarity": "none",
          "context": "Claude Haiku 4.5"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-cost-per-call",
          "series": "Input (prompt, cache priced)",
          "point": "Claude Opus 5.5 · Claude Code",
          "metric": "thinking-bill-cost-per-call/Input (prompt, cache priced)",
          "label": "List-price cost per call: reasoning, remaining output and input (calculation) (Input (prompt, cache priced))",
          "value": 0.00771,
          "unit": "usd",
          "display": "$0.0077",
          "n": 24,
          "calculation": true,
          "polarity": "none",
          "context": "Claude Opus 5.5"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-cost-per-call",
          "series": "Input (prompt, cache priced)",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "thinking-bill-cost-per-call/Input (prompt, cache priced)",
          "label": "List-price cost per call: reasoning, remaining output and input (calculation) (Input (prompt, cache priced))",
          "value": 0.004012,
          "unit": "usd",
          "display": "$0.0040",
          "n": 24,
          "calculation": true,
          "polarity": "none",
          "context": "Claude Sonnet 5.5"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-by-effort",
          "series": "Reasoning cost per strict pass",
          "point": "Claude Sonnet 5.5 (low) · Claude Code",
          "metric": "thinking-bill-by-effort/Reasoning cost per strict pass",
          "label": "Reasoning cost per strict pass by effort, with the total (calculation) (Reasoning cost per strict pass)",
          "value": 0.004317,
          "unit": "usd",
          "display": "$0.0043",
          "n": 16,
          "calculation": true,
          "polarity": "none",
          "context": "Claude Sonnet 5.5 · effort low"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-by-effort",
          "series": "Reasoning cost per strict pass",
          "point": "Claude Sonnet 5.5 (medium) · Claude Code",
          "metric": "thinking-bill-by-effort/Reasoning cost per strict pass",
          "label": "Reasoning cost per strict pass by effort, with the total (calculation) (Reasoning cost per strict pass)",
          "value": 0.005946,
          "unit": "usd",
          "display": "$0.0059",
          "n": 16,
          "calculation": true,
          "polarity": "none",
          "context": "Claude Sonnet 5.5 · effort medium"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-by-effort",
          "series": "Reasoning cost per strict pass",
          "point": "Claude Sonnet 5.5 (high) · Claude Code",
          "metric": "thinking-bill-by-effort/Reasoning cost per strict pass",
          "label": "Reasoning cost per strict pass by effort, with the total (calculation) (Reasoning cost per strict pass)",
          "value": 0.009369,
          "unit": "usd",
          "display": "$0.0094",
          "n": 16,
          "calculation": true,
          "polarity": "none",
          "context": "Claude Sonnet 5.5 · effort high"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-by-effort",
          "series": "Reasoning cost per strict pass",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "thinking-bill-by-effort/Reasoning cost per strict pass",
          "label": "Reasoning cost per strict pass by effort, with the total (calculation) (Reasoning cost per strict pass)",
          "value": 0.006299,
          "unit": "usd",
          "display": "$0.0063",
          "n": 16,
          "calculation": true,
          "polarity": "none",
          "context": "Claude Sonnet 5.5"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-by-effort",
          "series": "Reasoning cost per strict pass",
          "point": "Claude Opus 5.5 (low) · Claude Code",
          "metric": "thinking-bill-by-effort/Reasoning cost per strict pass",
          "label": "Reasoning cost per strict pass by effort, with the total (calculation) (Reasoning cost per strict pass)",
          "value": 0.005031,
          "unit": "usd",
          "display": "$0.0050",
          "n": 16,
          "calculation": true,
          "polarity": "none",
          "context": "Claude Opus 5.5 · effort low"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-by-effort",
          "series": "Reasoning cost per strict pass",
          "point": "Claude Opus 5.5 (medium) · Claude Code",
          "metric": "thinking-bill-by-effort/Reasoning cost per strict pass",
          "label": "Reasoning cost per strict pass by effort, with the total (calculation) (Reasoning cost per strict pass)",
          "value": 0.01344,
          "unit": "usd",
          "display": "$0.013",
          "n": 16,
          "calculation": true,
          "polarity": "none",
          "context": "Claude Opus 5.5 · effort medium"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-by-effort",
          "series": "Reasoning cost per strict pass",
          "point": "Claude Opus 5.5 (high) · Claude Code",
          "metric": "thinking-bill-by-effort/Reasoning cost per strict pass",
          "label": "Reasoning cost per strict pass by effort, with the total (calculation) (Reasoning cost per strict pass)",
          "value": 0.018034,
          "unit": "usd",
          "display": "$0.018",
          "n": 16,
          "calculation": true,
          "polarity": "none",
          "context": "Claude Opus 5.5 · effort high"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-by-effort",
          "series": "Reasoning cost per strict pass",
          "point": "Claude Opus 5.5 · Claude Code",
          "metric": "thinking-bill-by-effort/Reasoning cost per strict pass",
          "label": "Reasoning cost per strict pass by effort, with the total (calculation) (Reasoning cost per strict pass)",
          "value": 0.013104,
          "unit": "usd",
          "display": "$0.013",
          "n": 16,
          "calculation": true,
          "polarity": "none",
          "context": "Claude Opus 5.5"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-by-effort",
          "series": "Total cost per strict pass",
          "point": "Claude Sonnet 5.5 (low) · Claude Code",
          "metric": "thinking-bill-by-effort/Total cost per strict pass",
          "label": "Reasoning cost per strict pass by effort, with the total (calculation) (Total cost per strict pass)",
          "value": 0.012191,
          "unit": "usd",
          "display": "$0.012",
          "n": 16,
          "calculation": true,
          "polarity": "none",
          "context": "Claude Sonnet 5.5 · effort low"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-by-effort",
          "series": "Total cost per strict pass",
          "point": "Claude Sonnet 5.5 (medium) · Claude Code",
          "metric": "thinking-bill-by-effort/Total cost per strict pass",
          "label": "Reasoning cost per strict pass by effort, with the total (calculation) (Total cost per strict pass)",
          "value": 0.01352,
          "unit": "usd",
          "display": "$0.014",
          "n": 16,
          "calculation": true,
          "polarity": "none",
          "context": "Claude Sonnet 5.5 · effort medium"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-by-effort",
          "series": "Total cost per strict pass",
          "point": "Claude Sonnet 5.5 (high) · Claude Code",
          "metric": "thinking-bill-by-effort/Total cost per strict pass",
          "label": "Reasoning cost per strict pass by effort, with the total (calculation) (Total cost per strict pass)",
          "value": 0.016705,
          "unit": "usd",
          "display": "$0.017",
          "n": 16,
          "calculation": true,
          "polarity": "none",
          "context": "Claude Sonnet 5.5 · effort high"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-by-effort",
          "series": "Total cost per strict pass",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "thinking-bill-by-effort/Total cost per strict pass",
          "label": "Reasoning cost per strict pass by effort, with the total (calculation) (Total cost per strict pass)",
          "value": 0.013978,
          "unit": "usd",
          "display": "$0.014",
          "n": 16,
          "calculation": true,
          "polarity": "none",
          "context": "Claude Sonnet 5.5"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-by-effort",
          "series": "Total cost per strict pass",
          "point": "Claude Opus 5.5 (low) · Claude Code",
          "metric": "thinking-bill-by-effort/Total cost per strict pass",
          "label": "Reasoning cost per strict pass by effort, with the total (calculation) (Total cost per strict pass)",
          "value": 0.021152,
          "unit": "usd",
          "display": "$0.021",
          "n": 16,
          "calculation": true,
          "polarity": "none",
          "context": "Claude Opus 5.5 · effort low"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-by-effort",
          "series": "Total cost per strict pass",
          "point": "Claude Opus 5.5 (medium) · Claude Code",
          "metric": "thinking-bill-by-effort/Total cost per strict pass",
          "label": "Reasoning cost per strict pass by effort, with the total (calculation) (Total cost per strict pass)",
          "value": 0.029475,
          "unit": "usd",
          "display": "$0.029",
          "n": 16,
          "calculation": true,
          "polarity": "none",
          "context": "Claude Opus 5.5 · effort medium"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-by-effort",
          "series": "Total cost per strict pass",
          "point": "Claude Opus 5.5 (high) · Claude Code",
          "metric": "thinking-bill-by-effort/Total cost per strict pass",
          "label": "Reasoning cost per strict pass by effort, with the total (calculation) (Total cost per strict pass)",
          "value": 0.033677,
          "unit": "usd",
          "display": "$0.034",
          "n": 16,
          "calculation": true,
          "polarity": "none",
          "context": "Claude Opus 5.5 · effort high"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-by-effort",
          "series": "Total cost per strict pass",
          "point": "Claude Opus 5.5 · Claude Code",
          "metric": "thinking-bill-by-effort/Total cost per strict pass",
          "label": "Reasoning cost per strict pass by effort, with the total (calculation) (Total cost per strict pass)",
          "value": 0.028925,
          "unit": "usd",
          "display": "$0.029",
          "n": 16,
          "calculation": true,
          "polarity": "none",
          "context": "Claude Opus 5.5"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-short-vs-hard",
          "series": "Eight hard tasks",
          "point": "Claude Haiku 4.5 · Claude Code",
          "metric": "thinking-bill-short-vs-hard/Eight hard tasks",
          "label": "Reasoning share on short tasks vs hard tasks (calculation) (Eight hard tasks)",
          "value": 91.68,
          "unit": "percent",
          "display": "91.7%",
          "n": 24,
          "range": [
            76.46,
            99.27
          ],
          "spanKind": "minmax",
          "calculation": true,
          "polarity": "none",
          "context": "Claude Haiku 4.5"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-short-vs-hard",
          "series": "Eight hard tasks",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "thinking-bill-short-vs-hard/Eight hard tasks",
          "label": "Reasoning share on short tasks vs hard tasks (calculation) (Eight hard tasks)",
          "value": 54.54,
          "unit": "percent",
          "display": "54.5%",
          "n": 24,
          "range": [
            0,
            95.91
          ],
          "spanKind": "minmax",
          "calculation": true,
          "polarity": "none",
          "context": "Claude Sonnet 5.5"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-short-vs-hard",
          "series": "Eight hard tasks",
          "point": "Claude Opus 5.5 · Claude Code",
          "metric": "thinking-bill-short-vs-hard/Eight hard tasks",
          "label": "Reasoning share on short tasks vs hard tasks (calculation) (Eight hard tasks)",
          "value": 54.79,
          "unit": "percent",
          "display": "54.8%",
          "n": 24,
          "range": [
            29.92,
            95.6
          ],
          "spanKind": "minmax",
          "calculation": true,
          "polarity": "none",
          "context": "Claude Opus 5.5"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-short-vs-hard",
          "series": "Eight hard tasks",
          "point": "Claude Opus 5.5 (high) · Claude Code",
          "metric": "thinking-bill-short-vs-hard/Eight hard tasks",
          "label": "Reasoning share on short tasks vs hard tasks (calculation) (Eight hard tasks)",
          "value": 54.43,
          "unit": "percent",
          "display": "54.4%",
          "n": 24,
          "range": [
            36.14,
            96.23
          ],
          "spanKind": "minmax",
          "calculation": true,
          "polarity": "none",
          "context": "Claude Opus 5.5 · effort high"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-short-vs-hard",
          "series": "Eight hard tasks",
          "point": "Claude Fable 5.1 · Claude Code",
          "metric": "thinking-bill-short-vs-hard/Eight hard tasks",
          "label": "Reasoning share on short tasks vs hard tasks (calculation) (Eight hard tasks)",
          "value": 64.24,
          "unit": "percent",
          "display": "64.2%",
          "n": 24,
          "range": [
            23.44,
            97.19
          ],
          "spanKind": "minmax",
          "calculation": true,
          "polarity": "none",
          "context": "Claude Fable 5.1"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-short-vs-hard",
          "series": "Five short tasks",
          "point": "Claude Haiku 4.5 · Claude Code",
          "metric": "thinking-bill-short-vs-hard/Five short tasks",
          "label": "Reasoning share on short tasks vs hard tasks (calculation) (Five short tasks)",
          "value": 90.19,
          "unit": "percent",
          "display": "90.2%",
          "n": 15,
          "range": [
            73.1,
            97.59
          ],
          "spanKind": "minmax",
          "calculation": true,
          "polarity": "none",
          "context": "Claude Haiku 4.5"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-short-vs-hard",
          "series": "Five short tasks",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "thinking-bill-short-vs-hard/Five short tasks",
          "label": "Reasoning share on short tasks vs hard tasks (calculation) (Five short tasks)",
          "value": 0,
          "unit": "percent",
          "display": "0%",
          "n": 15,
          "range": [
            0,
            72.75
          ],
          "spanKind": "minmax",
          "calculation": true,
          "polarity": "none",
          "context": "Claude Sonnet 5.5"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-short-vs-hard",
          "series": "Five short tasks",
          "point": "Claude Opus 5.5 · Claude Code",
          "metric": "thinking-bill-short-vs-hard/Five short tasks",
          "label": "Reasoning share on short tasks vs hard tasks (calculation) (Five short tasks)",
          "value": 0,
          "unit": "percent",
          "display": "0%",
          "n": 15,
          "range": [
            0,
            93.33
          ],
          "spanKind": "minmax",
          "calculation": true,
          "polarity": "none",
          "context": "Claude Opus 5.5"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-short-vs-hard",
          "series": "Five short tasks",
          "point": "Claude Opus 5.5 (high) · Claude Code",
          "metric": "thinking-bill-short-vs-hard/Five short tasks",
          "label": "Reasoning share on short tasks vs hard tasks (calculation) (Five short tasks)",
          "value": 43.59,
          "unit": "percent",
          "display": "43.6%",
          "n": 15,
          "range": [
            0,
            93.33
          ],
          "spanKind": "minmax",
          "calculation": true,
          "polarity": "none",
          "context": "Claude Opus 5.5 · effort high"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-short-vs-hard",
          "series": "Five short tasks",
          "point": "Claude Fable 5.1 · Claude Code",
          "metric": "thinking-bill-short-vs-hard/Five short tasks",
          "label": "Reasoning share on short tasks vs hard tasks (calculation) (Five short tasks)",
          "value": 0,
          "unit": "percent",
          "display": "0%",
          "n": 15,
          "range": [
            0,
            74.01
          ],
          "spanKind": "minmax",
          "calculation": true,
          "polarity": "none",
          "context": "Claude Fable 5.1"
        },
        {
          "studySlug": "llm-speed-anatomy",
          "chartId": "speed-anatomy-first-text",
          "series": "Time to first text",
          "point": "Claude Haiku 4.5 · Claude Code",
          "metric": "speed-anatomy-first-text",
          "label": "Time to first text: a 250-line answer, six models",
          "value": 4,
          "unit": "seconds",
          "display": "4.00 s",
          "n": 4,
          "range": [
            2.84,
            6.38
          ],
          "spanKind": "minmax",
          "context": "Claude Haiku 4.5"
        },
        {
          "studySlug": "llm-speed-anatomy",
          "chartId": "speed-anatomy-first-text",
          "series": "Time to first text",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "speed-anatomy-first-text",
          "label": "Time to first text: a 250-line answer, six models",
          "value": 1.96,
          "unit": "seconds",
          "display": "1.96 s",
          "n": 4,
          "range": [
            0.88,
            4.09
          ],
          "spanKind": "minmax",
          "context": "Claude Sonnet 5.5"
        },
        {
          "studySlug": "llm-speed-anatomy",
          "chartId": "speed-anatomy-first-text",
          "series": "Time to first text",
          "point": "Claude Opus 5.5 · Claude Code",
          "metric": "speed-anatomy-first-text",
          "label": "Time to first text: a 250-line answer, six models",
          "value": 1.97,
          "unit": "seconds",
          "display": "1.97 s",
          "n": 4,
          "range": [
            1.7,
            2.35
          ],
          "spanKind": "minmax",
          "context": "Claude Opus 5.5"
        },
        {
          "studySlug": "llm-speed-anatomy",
          "chartId": "speed-anatomy-first-text",
          "series": "Time to first text",
          "point": "Claude Fable 5.1 · Claude Code",
          "metric": "speed-anatomy-first-text",
          "label": "Time to first text: a 250-line answer, six models",
          "value": 4.43,
          "unit": "seconds",
          "display": "4.43 s",
          "n": 4,
          "range": [
            2.27,
            4.64
          ],
          "spanKind": "minmax",
          "context": "Claude Fable 5.1"
        },
        {
          "studySlug": "llm-speed-anatomy",
          "chartId": "speed-anatomy-output-speed",
          "series": "Visible tokens per second",
          "point": "Claude Haiku 4.5 · Claude Code",
          "metric": "speed-anatomy-output-speed",
          "label": "Output speed after the first text: visible tokens per second (calculation)",
          "value": 153.2,
          "unit": "tokens",
          "display": "153",
          "n": 4,
          "range": [
            152.6,
            216.1
          ],
          "spanKind": "minmax",
          "calculation": true,
          "context": "Claude Haiku 4.5"
        },
        {
          "studySlug": "llm-speed-anatomy",
          "chartId": "speed-anatomy-output-speed",
          "series": "Visible tokens per second",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "speed-anatomy-output-speed",
          "label": "Output speed after the first text: visible tokens per second (calculation)",
          "value": 231.7,
          "unit": "tokens",
          "display": "232",
          "n": 4,
          "range": [
            230.3,
            233
          ],
          "spanKind": "minmax",
          "calculation": true,
          "context": "Claude Sonnet 5.5"
        },
        {
          "studySlug": "llm-speed-anatomy",
          "chartId": "speed-anatomy-output-speed",
          "series": "Visible tokens per second",
          "point": "Claude Opus 5.5 · Claude Code",
          "metric": "speed-anatomy-output-speed",
          "label": "Output speed after the first text: visible tokens per second (calculation)",
          "value": 155.5,
          "unit": "tokens",
          "display": "156",
          "n": 4,
          "range": [
            154.6,
            156.4
          ],
          "spanKind": "minmax",
          "calculation": true,
          "context": "Claude Opus 5.5"
        },
        {
          "studySlug": "llm-speed-anatomy",
          "chartId": "speed-anatomy-output-speed",
          "series": "Visible tokens per second",
          "point": "Claude Fable 5.1 · Claude Code",
          "metric": "speed-anatomy-output-speed",
          "label": "Output speed after the first text: visible tokens per second (calculation)",
          "value": 122.6,
          "unit": "tokens",
          "display": "123",
          "n": 4,
          "range": [
            120.9,
            131.4
          ],
          "spanKind": "minmax",
          "calculation": true,
          "context": "Claude Fable 5.1"
        },
        {
          "studySlug": "llm-speed-anatomy",
          "chartId": "speed-anatomy-chars-per-second",
          "series": "Characters per second",
          "point": "Claude Haiku 4.5 · Claude Code",
          "metric": "speed-anatomy-chars-per-second",
          "label": "Output speed in characters per second after the first text (calculation)",
          "value": 547,
          "unit": "count",
          "display": "547",
          "n": 3,
          "range": [
            546,
            548
          ],
          "spanKind": "minmax",
          "calculation": true,
          "context": "Claude Haiku 4.5"
        },
        {
          "studySlug": "llm-speed-anatomy",
          "chartId": "speed-anatomy-chars-per-second",
          "series": "Characters per second",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "speed-anatomy-chars-per-second",
          "label": "Output speed in characters per second after the first text (calculation)",
          "value": 517,
          "unit": "count",
          "display": "517",
          "n": 4,
          "range": [
            513,
            519
          ],
          "spanKind": "minmax",
          "calculation": true,
          "context": "Claude Sonnet 5.5"
        },
        {
          "studySlug": "llm-speed-anatomy",
          "chartId": "speed-anatomy-chars-per-second",
          "series": "Characters per second",
          "point": "Claude Opus 5.5 · Claude Code",
          "metric": "speed-anatomy-chars-per-second",
          "label": "Output speed in characters per second after the first text (calculation)",
          "value": 347,
          "unit": "count",
          "display": "347",
          "n": 4,
          "range": [
            345,
            349
          ],
          "spanKind": "minmax",
          "calculation": true,
          "context": "Claude Opus 5.5"
        },
        {
          "studySlug": "llm-speed-anatomy",
          "chartId": "speed-anatomy-chars-per-second",
          "series": "Characters per second",
          "point": "Claude Fable 5.1 · Claude Code",
          "metric": "speed-anatomy-chars-per-second",
          "label": "Output speed in characters per second after the first text (calculation)",
          "value": 273,
          "unit": "count",
          "display": "273",
          "n": 4,
          "range": [
            270,
            293
          ],
          "spanKind": "minmax",
          "calculation": true,
          "context": "Claude Fable 5.1"
        },
        {
          "studySlug": "llm-speed-anatomy",
          "chartId": "speed-anatomy-prompt-size",
          "series": "Claude Haiku 4.5 · Claude Code",
          "point": "1k",
          "metric": "speed-anatomy-prompt-size/1k",
          "label": "Time to first text as the prompt grows: 1k",
          "value": 1.93,
          "unit": "seconds",
          "display": "1.93 s",
          "n": 3,
          "range": [
            1.85,
            2.04
          ],
          "spanKind": "minmax",
          "calculation": true,
          "context": "Claude Haiku 4.5"
        },
        {
          "studySlug": "llm-speed-anatomy",
          "chartId": "speed-anatomy-prompt-size",
          "series": "Claude Haiku 4.5 · Claude Code",
          "point": "16k",
          "metric": "speed-anatomy-prompt-size/16k",
          "label": "Time to first text as the prompt grows: 16k",
          "value": 2.27,
          "unit": "seconds",
          "display": "2.27 s",
          "n": 3,
          "range": [
            2.22,
            2.47
          ],
          "spanKind": "minmax",
          "calculation": true,
          "context": "Claude Haiku 4.5"
        },
        {
          "studySlug": "llm-speed-anatomy",
          "chartId": "speed-anatomy-prompt-size",
          "series": "Claude Haiku 4.5 · Claude Code",
          "point": "64k",
          "metric": "speed-anatomy-prompt-size/64k",
          "label": "Time to first text as the prompt grows: 64k",
          "value": 2.78,
          "unit": "seconds",
          "display": "2.78 s",
          "n": 3,
          "range": [
            2.45,
            2.89
          ],
          "spanKind": "minmax",
          "calculation": true,
          "context": "Claude Haiku 4.5"
        },
        {
          "studySlug": "llm-speed-anatomy",
          "chartId": "speed-anatomy-prompt-size",
          "series": "Claude Sonnet 5.5 · Claude Code",
          "point": "1k",
          "metric": "speed-anatomy-prompt-size/1k",
          "label": "Time to first text as the prompt grows: 1k",
          "value": 1.45,
          "unit": "seconds",
          "display": "1.45 s",
          "n": 3,
          "range": [
            1.23,
            1.72
          ],
          "spanKind": "minmax",
          "calculation": true,
          "context": "Claude Sonnet 5.5"
        },
        {
          "studySlug": "llm-speed-anatomy",
          "chartId": "speed-anatomy-prompt-size",
          "series": "Claude Sonnet 5.5 · Claude Code",
          "point": "16k",
          "metric": "speed-anatomy-prompt-size/16k",
          "label": "Time to first text as the prompt grows: 16k",
          "value": 1.78,
          "unit": "seconds",
          "display": "1.78 s",
          "n": 3,
          "range": [
            1.64,
            2.11
          ],
          "spanKind": "minmax",
          "calculation": true,
          "context": "Claude Sonnet 5.5"
        },
        {
          "studySlug": "llm-speed-anatomy",
          "chartId": "speed-anatomy-prompt-size",
          "series": "Claude Sonnet 5.5 · Claude Code",
          "point": "64k",
          "metric": "speed-anatomy-prompt-size/64k",
          "label": "Time to first text as the prompt grows: 64k",
          "value": 3.07,
          "unit": "seconds",
          "display": "3.07 s",
          "n": 3,
          "range": [
            1.38,
            3.61
          ],
          "spanKind": "minmax",
          "calculation": true,
          "context": "Claude Sonnet 5.5"
        },
        {
          "studySlug": "llm-speed-anatomy",
          "chartId": "speed-anatomy-prompt-size",
          "series": "Claude Opus 5.5 · Claude Code",
          "point": "1k",
          "metric": "speed-anatomy-prompt-size/1k",
          "label": "Time to first text as the prompt grows: 1k",
          "value": 1.51,
          "unit": "seconds",
          "display": "1.51 s",
          "n": 3,
          "range": [
            1.46,
            2.01
          ],
          "spanKind": "minmax",
          "calculation": true,
          "context": "Claude Opus 5.5"
        },
        {
          "studySlug": "llm-speed-anatomy",
          "chartId": "speed-anatomy-prompt-size",
          "series": "Claude Opus 5.5 · Claude Code",
          "point": "16k",
          "metric": "speed-anatomy-prompt-size/16k",
          "label": "Time to first text as the prompt grows: 16k",
          "value": 1.74,
          "unit": "seconds",
          "display": "1.74 s",
          "n": 3,
          "range": [
            1.7,
            2.97
          ],
          "spanKind": "minmax",
          "calculation": true,
          "context": "Claude Opus 5.5"
        },
        {
          "studySlug": "llm-speed-anatomy",
          "chartId": "speed-anatomy-prompt-size",
          "series": "Claude Opus 5.5 · Claude Code",
          "point": "64k",
          "metric": "speed-anatomy-prompt-size/64k",
          "label": "Time to first text as the prompt grows: 64k",
          "value": 1.79,
          "unit": "seconds",
          "display": "1.79 s",
          "n": 3,
          "range": [
            1.72,
            3.72
          ],
          "spanKind": "minmax",
          "calculation": true,
          "context": "Claude Opus 5.5"
        },
        {
          "studySlug": "llm-speed-anatomy",
          "chartId": "speed-anatomy-total-by-size",
          "series": "1k prompt",
          "point": "Claude Haiku 4.5 · Claude Code",
          "metric": "speed-anatomy-total-by-size/1k prompt",
          "label": "Total time per call by prompt size (1k prompt)",
          "value": 2.34,
          "unit": "seconds",
          "display": "2.34 s",
          "n": 3,
          "range": [
            2.22,
            2.46
          ],
          "spanKind": "minmax",
          "context": "Claude Haiku 4.5"
        },
        {
          "studySlug": "llm-speed-anatomy",
          "chartId": "speed-anatomy-total-by-size",
          "series": "1k prompt",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "speed-anatomy-total-by-size/1k prompt",
          "label": "Total time per call by prompt size (1k prompt)",
          "value": 1.78,
          "unit": "seconds",
          "display": "1.78 s",
          "n": 3,
          "range": [
            1.57,
            2.12
          ],
          "spanKind": "minmax",
          "context": "Claude Sonnet 5.5"
        },
        {
          "studySlug": "llm-speed-anatomy",
          "chartId": "speed-anatomy-total-by-size",
          "series": "1k prompt",
          "point": "Claude Opus 5.5 · Claude Code",
          "metric": "speed-anatomy-total-by-size/1k prompt",
          "label": "Total time per call by prompt size (1k prompt)",
          "value": 1.83,
          "unit": "seconds",
          "display": "1.83 s",
          "n": 3,
          "range": [
            1.82,
            2.41
          ],
          "spanKind": "minmax",
          "context": "Claude Opus 5.5"
        },
        {
          "studySlug": "llm-speed-anatomy",
          "chartId": "speed-anatomy-total-by-size",
          "series": "16k prompt",
          "point": "Claude Haiku 4.5 · Claude Code",
          "metric": "speed-anatomy-total-by-size/16k prompt",
          "label": "Total time per call by prompt size (16k prompt)",
          "value": 2.79,
          "unit": "seconds",
          "display": "2.79 s",
          "n": 3,
          "range": [
            2.58,
            2.84
          ],
          "spanKind": "minmax",
          "context": "Claude Haiku 4.5"
        },
        {
          "studySlug": "llm-speed-anatomy",
          "chartId": "speed-anatomy-total-by-size",
          "series": "16k prompt",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "speed-anatomy-total-by-size/16k prompt",
          "label": "Total time per call by prompt size (16k prompt)",
          "value": 2.1,
          "unit": "seconds",
          "display": "2.10 s",
          "n": 3,
          "range": [
            1.98,
            2.48
          ],
          "spanKind": "minmax",
          "context": "Claude Sonnet 5.5"
        },
        {
          "studySlug": "llm-speed-anatomy",
          "chartId": "speed-anatomy-total-by-size",
          "series": "16k prompt",
          "point": "Claude Opus 5.5 · Claude Code",
          "metric": "speed-anatomy-total-by-size/16k prompt",
          "label": "Total time per call by prompt size (16k prompt)",
          "value": 2.36,
          "unit": "seconds",
          "display": "2.36 s",
          "n": 3,
          "range": [
            2.11,
            3.4
          ],
          "spanKind": "minmax",
          "context": "Claude Opus 5.5"
        },
        {
          "studySlug": "llm-speed-anatomy",
          "chartId": "speed-anatomy-total-by-size",
          "series": "64k prompt",
          "point": "Claude Haiku 4.5 · Claude Code",
          "metric": "speed-anatomy-total-by-size/64k prompt",
          "label": "Total time per call by prompt size (64k prompt)",
          "value": 3.13,
          "unit": "seconds",
          "display": "3.13 s",
          "n": 3,
          "range": [
            2.84,
            3.28
          ],
          "spanKind": "minmax",
          "context": "Claude Haiku 4.5"
        },
        {
          "studySlug": "llm-speed-anatomy",
          "chartId": "speed-anatomy-total-by-size",
          "series": "64k prompt",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "speed-anatomy-total-by-size/64k prompt",
          "label": "Total time per call by prompt size (64k prompt)",
          "value": 3.44,
          "unit": "seconds",
          "display": "3.44 s",
          "n": 3,
          "range": [
            1.74,
            4.38
          ],
          "spanKind": "minmax",
          "context": "Claude Sonnet 5.5"
        },
        {
          "studySlug": "llm-speed-anatomy",
          "chartId": "speed-anatomy-total-by-size",
          "series": "64k prompt",
          "point": "Claude Opus 5.5 · Claude Code",
          "metric": "speed-anatomy-total-by-size/64k prompt",
          "label": "Total time per call by prompt size (64k prompt)",
          "value": 2.35,
          "unit": "seconds",
          "display": "2.35 s",
          "n": 3,
          "range": [
            2.26,
            4.29
          ],
          "spanKind": "minmax",
          "context": "Claude Opus 5.5"
        },
        {
          "studySlug": "llm-speed-anatomy",
          "chartId": "speed-anatomy-lookup-correct",
          "series": "Exact answer",
          "point": "Claude Haiku 4.5 · Claude Code",
          "metric": "speed-anatomy-lookup-correct",
          "label": "Exact lookup answers at the 1k, 16k and 64k prompt-size targets",
          "value": 1,
          "unit": "rate",
          "display": "100% (9/9)",
          "n": 9,
          "ci": [
            0.7009,
            1
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Haiku 4.5"
        },
        {
          "studySlug": "llm-speed-anatomy",
          "chartId": "speed-anatomy-lookup-correct",
          "series": "Exact answer",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "speed-anatomy-lookup-correct",
          "label": "Exact lookup answers at the 1k, 16k and 64k prompt-size targets",
          "value": 1,
          "unit": "rate",
          "display": "100% (9/9)",
          "n": 9,
          "ci": [
            0.7009,
            1
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Sonnet 5.5"
        },
        {
          "studySlug": "llm-speed-anatomy",
          "chartId": "speed-anatomy-lookup-correct",
          "series": "Exact answer",
          "point": "Claude Opus 5.5 · Claude Code",
          "metric": "speed-anatomy-lookup-correct",
          "label": "Exact lookup answers at the 1k, 16k and 64k prompt-size targets",
          "value": 0.5556,
          "unit": "rate",
          "display": "56% (5/9)",
          "n": 9,
          "ci": [
            0.2667,
            0.8112
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Opus 5.5"
        },
        {
          "studySlug": "haiku-retry-or-escalate",
          "chartId": "retry-escalate-call-cost-by-task",
          "series": "Claude Haiku 4.5 · Claude Code",
          "point": "Interval merge fix",
          "metric": "retry-escalate-call-cost-by-task/Interval merge fix",
          "label": "List-price cost of one call, Haiku 4.5 vs Sonnet 5.5, on each hard task (calculation): Interval merge fix",
          "value": 0.01789,
          "unit": "usd",
          "display": "$0.018",
          "n": 3,
          "calculation": true,
          "context": "Claude Haiku 4.5"
        },
        {
          "studySlug": "haiku-retry-or-escalate",
          "chartId": "retry-escalate-call-cost-by-task",
          "series": "Claude Haiku 4.5 · Claude Code",
          "point": "DST day length",
          "metric": "retry-escalate-call-cost-by-task/DST day length",
          "label": "List-price cost of one call, Haiku 4.5 vs Sonnet 5.5, on each hard task (calculation): DST day length",
          "value": 0.0293,
          "unit": "usd",
          "display": "$0.029",
          "n": 3,
          "calculation": true,
          "context": "Claude Haiku 4.5"
        },
        {
          "studySlug": "haiku-retry-or-escalate",
          "chartId": "retry-escalate-call-cost-by-task",
          "series": "Claude Haiku 4.5 · Claude Code",
          "point": "CSV parser",
          "metric": "retry-escalate-call-cost-by-task/CSV parser",
          "label": "List-price cost of one call, Haiku 4.5 vs Sonnet 5.5, on each hard task (calculation): CSV parser",
          "value": 0.02919,
          "unit": "usd",
          "display": "$0.029",
          "n": 3,
          "calculation": true,
          "context": "Claude Haiku 4.5"
        },
        {
          "studySlug": "haiku-retry-or-escalate",
          "chartId": "retry-escalate-call-cost-by-task",
          "series": "Claude Haiku 4.5 · Claude Code",
          "point": "Event-loop order",
          "metric": "retry-escalate-call-cost-by-task/Event-loop order",
          "label": "List-price cost of one call, Haiku 4.5 vs Sonnet 5.5, on each hard task (calculation): Event-loop order",
          "value": 0.03708,
          "unit": "usd",
          "display": "$0.037",
          "n": 3,
          "calculation": true,
          "context": "Claude Haiku 4.5"
        },
        {
          "studySlug": "haiku-retry-or-escalate",
          "chartId": "retry-escalate-call-cost-by-task",
          "series": "Claude Haiku 4.5 · Claude Code",
          "point": "Room schedule",
          "metric": "retry-escalate-call-cost-by-task/Room schedule",
          "label": "List-price cost of one call, Haiku 4.5 vs Sonnet 5.5, on each hard task (calculation): Room schedule",
          "value": 0.03578,
          "unit": "usd",
          "display": "$0.036",
          "n": 3,
          "calculation": true,
          "context": "Claude Haiku 4.5"
        },
        {
          "studySlug": "haiku-retry-or-escalate",
          "chartId": "retry-escalate-call-cost-by-task",
          "series": "Claude Haiku 4.5 · Claude Code",
          "point": "SemVer regex",
          "metric": "retry-escalate-call-cost-by-task/SemVer regex",
          "label": "List-price cost of one call, Haiku 4.5 vs Sonnet 5.5, on each hard task (calculation): SemVer regex",
          "value": 0.04217,
          "unit": "usd",
          "display": "$0.042",
          "n": 3,
          "calculation": true,
          "context": "Claude Haiku 4.5"
        },
        {
          "studySlug": "haiku-retry-or-escalate",
          "chartId": "retry-escalate-call-cost-by-task",
          "series": "Claude Haiku 4.5 · Claude Code",
          "point": "Money refactor",
          "metric": "retry-escalate-call-cost-by-task/Money refactor",
          "label": "List-price cost of one call, Haiku 4.5 vs Sonnet 5.5, on each hard task (calculation): Money refactor",
          "value": 0.02039,
          "unit": "usd",
          "display": "$0.020",
          "n": 3,
          "calculation": true,
          "context": "Claude Haiku 4.5"
        },
        {
          "studySlug": "haiku-retry-or-escalate",
          "chartId": "retry-escalate-call-cost-by-task",
          "series": "Claude Haiku 4.5 · Claude Code",
          "point": "SQLite report query",
          "metric": "retry-escalate-call-cost-by-task/SQLite report query",
          "label": "List-price cost of one call, Haiku 4.5 vs Sonnet 5.5, on each hard task (calculation): SQLite report query",
          "value": 0.03648,
          "unit": "usd",
          "display": "$0.036",
          "n": 3,
          "calculation": true,
          "context": "Claude Haiku 4.5"
        },
        {
          "studySlug": "haiku-retry-or-escalate",
          "chartId": "retry-escalate-call-cost-by-task",
          "series": "Claude Sonnet 5.5 · Claude Code",
          "point": "Interval merge fix",
          "metric": "retry-escalate-call-cost-by-task/Interval merge fix",
          "label": "List-price cost of one call, Haiku 4.5 vs Sonnet 5.5, on each hard task (calculation): Interval merge fix",
          "value": 0.00557,
          "unit": "usd",
          "display": "$0.0056",
          "n": 3,
          "calculation": true,
          "context": "Claude Sonnet 5.5"
        },
        {
          "studySlug": "haiku-retry-or-escalate",
          "chartId": "retry-escalate-call-cost-by-task",
          "series": "Claude Sonnet 5.5 · Claude Code",
          "point": "DST day length",
          "metric": "retry-escalate-call-cost-by-task/DST day length",
          "label": "List-price cost of one call, Haiku 4.5 vs Sonnet 5.5, on each hard task (calculation): DST day length",
          "value": 0.02532,
          "unit": "usd",
          "display": "$0.025",
          "n": 3,
          "calculation": true,
          "context": "Claude Sonnet 5.5"
        },
        {
          "studySlug": "haiku-retry-or-escalate",
          "chartId": "retry-escalate-call-cost-by-task",
          "series": "Claude Sonnet 5.5 · Claude Code",
          "point": "CSV parser",
          "metric": "retry-escalate-call-cost-by-task/CSV parser",
          "label": "List-price cost of one call, Haiku 4.5 vs Sonnet 5.5, on each hard task (calculation): CSV parser",
          "value": 0.01464,
          "unit": "usd",
          "display": "$0.015",
          "n": 3,
          "calculation": true,
          "context": "Claude Sonnet 5.5"
        },
        {
          "studySlug": "haiku-retry-or-escalate",
          "chartId": "retry-escalate-call-cost-by-task",
          "series": "Claude Sonnet 5.5 · Claude Code",
          "point": "Event-loop order",
          "metric": "retry-escalate-call-cost-by-task/Event-loop order",
          "label": "List-price cost of one call, Haiku 4.5 vs Sonnet 5.5, on each hard task (calculation): Event-loop order",
          "value": 0.01588,
          "unit": "usd",
          "display": "$0.016",
          "n": 3,
          "calculation": true,
          "context": "Claude Sonnet 5.5"
        },
        {
          "studySlug": "haiku-retry-or-escalate",
          "chartId": "retry-escalate-call-cost-by-task",
          "series": "Claude Sonnet 5.5 · Claude Code",
          "point": "Room schedule",
          "metric": "retry-escalate-call-cost-by-task/Room schedule",
          "label": "List-price cost of one call, Haiku 4.5 vs Sonnet 5.5, on each hard task (calculation): Room schedule",
          "value": 0.01243,
          "unit": "usd",
          "display": "$0.012",
          "n": 3,
          "calculation": true,
          "context": "Claude Sonnet 5.5"
        },
        {
          "studySlug": "haiku-retry-or-escalate",
          "chartId": "retry-escalate-call-cost-by-task",
          "series": "Claude Sonnet 5.5 · Claude Code",
          "point": "SemVer regex",
          "metric": "retry-escalate-call-cost-by-task/SemVer regex",
          "label": "List-price cost of one call, Haiku 4.5 vs Sonnet 5.5, on each hard task (calculation): SemVer regex",
          "value": 0.00514,
          "unit": "usd",
          "display": "$0.0051",
          "n": 3,
          "calculation": true,
          "context": "Claude Sonnet 5.5"
        },
        {
          "studySlug": "haiku-retry-or-escalate",
          "chartId": "retry-escalate-call-cost-by-task",
          "series": "Claude Sonnet 5.5 · Claude Code",
          "point": "Money refactor",
          "metric": "retry-escalate-call-cost-by-task/Money refactor",
          "label": "List-price cost of one call, Haiku 4.5 vs Sonnet 5.5, on each hard task (calculation): Money refactor",
          "value": 0.00961,
          "unit": "usd",
          "display": "$0.0096",
          "n": 3,
          "calculation": true,
          "context": "Claude Sonnet 5.5"
        },
        {
          "studySlug": "haiku-retry-or-escalate",
          "chartId": "retry-escalate-call-cost-by-task",
          "series": "Claude Sonnet 5.5 · Claude Code",
          "point": "SQLite report query",
          "metric": "retry-escalate-call-cost-by-task/SQLite report query",
          "label": "List-price cost of one call, Haiku 4.5 vs Sonnet 5.5, on each hard task (calculation): SQLite report query",
          "value": 0.01719,
          "unit": "usd",
          "display": "$0.017",
          "n": 3,
          "calculation": true,
          "context": "Claude Sonnet 5.5"
        },
        {
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-pass-rate",
          "series": "Strict pass",
          "point": "Claude Opus 5.5 · Claude Code",
          "metric": "harder-h2h-pass-rate/Strict pass",
          "label": "Pass rate on 4 harder tasks (Strict pass)",
          "value": 0.4167,
          "unit": "rate",
          "display": "42% (5/12)",
          "n": 12,
          "ci": [
            0.1933,
            0.6805
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Opus 5.5"
        },
        {
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-pass-rate",
          "series": "Strict pass",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "harder-h2h-pass-rate/Strict pass",
          "label": "Pass rate on 4 harder tasks (Strict pass)",
          "value": 0.375,
          "unit": "rate",
          "display": "38% (6/16)",
          "n": 16,
          "ci": [
            0.1848,
            0.6136
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Sonnet 5.5"
        },
        {
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-pass-rate",
          "series": "Strict pass",
          "point": "Claude Haiku 4.5 · Claude Code",
          "metric": "harder-h2h-pass-rate/Strict pass",
          "label": "Pass rate on 4 harder tasks (Strict pass)",
          "value": 0,
          "unit": "rate",
          "display": "0% (0/12)",
          "n": 12,
          "ci": [
            0,
            0.2425
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Haiku 4.5"
        },
        {
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-pass-rate",
          "series": "Lenient (format misses counted)",
          "point": "Claude Opus 5.5 · Claude Code",
          "metric": "harder-h2h-pass-rate/Lenient (format misses counted)",
          "label": "Pass rate on 4 harder tasks (Lenient (format misses counted))",
          "value": 0.5,
          "unit": "rate",
          "display": "50% (6/12)",
          "n": 12,
          "ci": [
            0.2538,
            0.7462
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Opus 5.5"
        },
        {
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-pass-rate",
          "series": "Lenient (format misses counted)",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "harder-h2h-pass-rate/Lenient (format misses counted)",
          "label": "Pass rate on 4 harder tasks (Lenient (format misses counted))",
          "value": 0.375,
          "unit": "rate",
          "display": "38% (6/16)",
          "n": 16,
          "ci": [
            0.1848,
            0.6136
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Sonnet 5.5"
        },
        {
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-pass-rate",
          "series": "Lenient (format misses counted)",
          "point": "Claude Haiku 4.5 · Claude Code",
          "metric": "harder-h2h-pass-rate/Lenient (format misses counted)",
          "label": "Pass rate on 4 harder tasks (Lenient (format misses counted))",
          "value": 0,
          "unit": "rate",
          "display": "0% (0/12)",
          "n": 12,
          "ci": [
            0,
            0.2425
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Haiku 4.5"
        },
        {
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-tool-attempts",
          "series": "Tool attempt",
          "point": "Claude Opus 5.5 · Claude Code",
          "metric": "harder-h2h-tool-attempts",
          "label": "Calls that tried a tool although tools were off",
          "value": 0.4167,
          "unit": "rate",
          "display": "42% (5/12)",
          "n": 12,
          "ci": [
            0.1933,
            0.6805
          ],
          "spanKind": "ci95",
          "polarity": "none",
          "context": "Claude Opus 5.5"
        },
        {
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-tool-attempts",
          "series": "Tool attempt",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "harder-h2h-tool-attempts",
          "label": "Calls that tried a tool although tools were off",
          "value": 0.3125,
          "unit": "rate",
          "display": "31% (5/16)",
          "n": 16,
          "ci": [
            0.1416,
            0.556
          ],
          "spanKind": "ci95",
          "polarity": "none",
          "context": "Claude Sonnet 5.5"
        },
        {
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-tool-attempts",
          "series": "Tool attempt",
          "point": "Claude Haiku 4.5 · Claude Code",
          "metric": "harder-h2h-tool-attempts",
          "label": "Calls that tried a tool although tools were off",
          "value": 0.0833,
          "unit": "rate",
          "display": "8% (1/12)",
          "n": 12,
          "ci": [
            0.0149,
            0.3539
          ],
          "spanKind": "ci95",
          "polarity": "none",
          "context": "Claude Haiku 4.5"
        },
        {
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-pass-by-task",
          "series": "Claude Opus 5.5 · Claude Code",
          "point": "10x10 nonogram",
          "metric": "harder-h2h-pass-by-task/10x10 nonogram",
          "label": "Strict pass rate by task: 10x10 nonogram",
          "value": 1,
          "unit": "rate",
          "display": "100% (3/3)",
          "n": 3,
          "ci": [
            0.4385,
            1
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Opus 5.5"
        },
        {
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-pass-by-task",
          "series": "Claude Opus 5.5 · Claude Code",
          "point": "Sudoku, 22 givens",
          "metric": "harder-h2h-pass-by-task/Sudoku, 22 givens",
          "label": "Strict pass rate by task: Sudoku, 22 givens",
          "value": 0,
          "unit": "rate",
          "display": "0% (0/3)",
          "n": 3,
          "ci": [
            0,
            0.5615
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Opus 5.5"
        },
        {
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-pass-by-task",
          "series": "Claude Opus 5.5 · Claude Code",
          "point": "6x6 Skyscrapers",
          "metric": "harder-h2h-pass-by-task/6x6 Skyscrapers",
          "label": "Strict pass rate by task: 6x6 Skyscrapers",
          "value": 0.3333,
          "unit": "rate",
          "display": "33% (1/3)",
          "n": 3,
          "ci": [
            0.0615,
            0.7923
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Opus 5.5"
        },
        {
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-pass-by-task",
          "series": "Claude Opus 5.5 · Claude Code",
          "point": "Seeded shuffle output",
          "metric": "harder-h2h-pass-by-task/Seeded shuffle output",
          "label": "Strict pass rate by task: Seeded shuffle output",
          "value": 0.3333,
          "unit": "rate",
          "display": "33% (1/3)",
          "n": 3,
          "ci": [
            0.0615,
            0.7923
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Opus 5.5"
        },
        {
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-pass-by-task",
          "series": "Claude Sonnet 5.5 · Claude Code",
          "point": "10x10 nonogram",
          "metric": "harder-h2h-pass-by-task/10x10 nonogram",
          "label": "Strict pass rate by task: 10x10 nonogram",
          "value": 1,
          "unit": "rate",
          "display": "100% (4/4)",
          "n": 4,
          "ci": [
            0.5101,
            1
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Sonnet 5.5"
        },
        {
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-pass-by-task",
          "series": "Claude Sonnet 5.5 · Claude Code",
          "point": "Sudoku, 22 givens",
          "metric": "harder-h2h-pass-by-task/Sudoku, 22 givens",
          "label": "Strict pass rate by task: Sudoku, 22 givens",
          "value": 0,
          "unit": "rate",
          "display": "0% (0/4)",
          "n": 4,
          "ci": [
            0,
            0.4899
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Sonnet 5.5"
        },
        {
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-pass-by-task",
          "series": "Claude Sonnet 5.5 · Claude Code",
          "point": "6x6 Skyscrapers",
          "metric": "harder-h2h-pass-by-task/6x6 Skyscrapers",
          "label": "Strict pass rate by task: 6x6 Skyscrapers",
          "value": 0,
          "unit": "rate",
          "display": "0% (0/4)",
          "n": 4,
          "ci": [
            0,
            0.4899
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Sonnet 5.5"
        },
        {
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-pass-by-task",
          "series": "Claude Sonnet 5.5 · Claude Code",
          "point": "Seeded shuffle output",
          "metric": "harder-h2h-pass-by-task/Seeded shuffle output",
          "label": "Strict pass rate by task: Seeded shuffle output",
          "value": 0.5,
          "unit": "rate",
          "display": "50% (2/4)",
          "n": 4,
          "ci": [
            0.15,
            0.85
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Sonnet 5.5"
        },
        {
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-pass-by-task",
          "series": "Claude Haiku 4.5 · Claude Code",
          "point": "10x10 nonogram",
          "metric": "harder-h2h-pass-by-task/10x10 nonogram",
          "label": "Strict pass rate by task: 10x10 nonogram",
          "value": 0,
          "unit": "rate",
          "display": "0% (0/3)",
          "n": 3,
          "ci": [
            0,
            0.5615
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Haiku 4.5"
        },
        {
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-pass-by-task",
          "series": "Claude Haiku 4.5 · Claude Code",
          "point": "Sudoku, 22 givens",
          "metric": "harder-h2h-pass-by-task/Sudoku, 22 givens",
          "label": "Strict pass rate by task: Sudoku, 22 givens",
          "value": 0,
          "unit": "rate",
          "display": "0% (0/3)",
          "n": 3,
          "ci": [
            0,
            0.5615
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Haiku 4.5"
        },
        {
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-pass-by-task",
          "series": "Claude Haiku 4.5 · Claude Code",
          "point": "6x6 Skyscrapers",
          "metric": "harder-h2h-pass-by-task/6x6 Skyscrapers",
          "label": "Strict pass rate by task: 6x6 Skyscrapers",
          "value": 0,
          "unit": "rate",
          "display": "0% (0/3)",
          "n": 3,
          "ci": [
            0,
            0.5615
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Haiku 4.5"
        },
        {
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-pass-by-task",
          "series": "Claude Haiku 4.5 · Claude Code",
          "point": "Seeded shuffle output",
          "metric": "harder-h2h-pass-by-task/Seeded shuffle output",
          "label": "Strict pass rate by task: Seeded shuffle output",
          "value": 0,
          "unit": "rate",
          "display": "0% (0/3)",
          "n": 3,
          "ci": [
            0,
            0.5615
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Claude Haiku 4.5"
        },
        {
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-total-latency",
          "series": "Total time per call",
          "point": "Claude Opus 5.5 · Claude Code",
          "metric": "harder-h2h-total-latency",
          "label": "Total time per call on harder tasks",
          "value": 80.34,
          "unit": "seconds",
          "display": "80.3 s",
          "n": 9,
          "range": [
            3.82,
            279.5
          ],
          "spanKind": "minmax",
          "context": "Claude Opus 5.5"
        },
        {
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-total-latency",
          "series": "Total time per call",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "harder-h2h-total-latency",
          "label": "Total time per call on harder tasks",
          "value": 70.43,
          "unit": "seconds",
          "display": "70.4 s",
          "n": 12,
          "range": [
            4.32,
            210.08
          ],
          "spanKind": "minmax",
          "context": "Claude Sonnet 5.5"
        },
        {
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-total-latency",
          "series": "Total time per call",
          "point": "Claude Haiku 4.5 · Claude Code",
          "metric": "harder-h2h-total-latency",
          "label": "Total time per call on harder tasks",
          "value": 108.98,
          "unit": "seconds",
          "display": "109.0 s",
          "n": 10,
          "range": [
            25.73,
            223.95
          ],
          "spanKind": "minmax",
          "context": "Claude Haiku 4.5"
        },
        {
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-output-tokens",
          "series": "Output tokens",
          "point": "Claude Opus 5.5 · Claude Code",
          "metric": "harder-h2h-output-tokens/Output tokens",
          "label": "Output tokens per call on harder tasks (Output tokens)",
          "value": 8420,
          "unit": "tokens",
          "display": "8,420",
          "n": 9,
          "range": [
            279,
            40044
          ],
          "spanKind": "minmax",
          "polarity": "none",
          "context": "Claude Opus 5.5"
        },
        {
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-output-tokens",
          "series": "Output tokens",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "harder-h2h-output-tokens/Output tokens",
          "label": "Output tokens per call on harder tasks (Output tokens)",
          "value": 9287,
          "unit": "tokens",
          "display": "9,287",
          "n": 12,
          "range": [
            407,
            27921
          ],
          "spanKind": "minmax",
          "polarity": "none",
          "context": "Claude Sonnet 5.5"
        },
        {
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-output-tokens",
          "series": "Output tokens",
          "point": "Claude Haiku 4.5 · Claude Code",
          "metric": "harder-h2h-output-tokens/Output tokens",
          "label": "Output tokens per call on harder tasks (Output tokens)",
          "value": 12508,
          "unit": "tokens",
          "display": "12,508",
          "n": 10,
          "range": [
            2965,
            26532
          ],
          "spanKind": "minmax",
          "polarity": "none",
          "context": "Claude Haiku 4.5"
        },
        {
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-cost-per-pass",
          "series": "Cost per strict pass",
          "point": "Claude Sonnet 5.5 · Claude Code",
          "metric": "harder-h2h-cost-per-pass",
          "label": "List-price cost per strict pass on harder tasks (calculation)",
          "value": 0.23843,
          "unit": "usd",
          "display": "$0.24",
          "n": 16,
          "calculation": true,
          "context": "Claude Sonnet 5.5"
        },
        {
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-cost-per-pass",
          "series": "Cost per strict pass",
          "point": "Claude Opus 5.5 · Claude Code",
          "metric": "harder-h2h-cost-per-pass",
          "label": "List-price cost per strict pass on harder tasks (calculation)",
          "value": 0.59333,
          "unit": "usd",
          "display": "$0.59",
          "n": 12,
          "calculation": true,
          "context": "Claude Opus 5.5"
        }
      ]
    },
    {
      "slug": "codex-cli",
      "name": "Codex CLI",
      "vendor": "OpenAI",
      "kind": "cli",
      "description": "OpenAI’s coding CLI. Each measurement pairs it with one GPT model and effort; the context names them.",
      "aliases": [
        "Codex CLI"
      ],
      "facts": [
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-pass-rate",
          "series": "Pass rate",
          "point": "GPT-6.1 Sol (high) · Codex CLI",
          "metric": "h2h-pass-rate",
          "label": "Pass rate on five validated tasks",
          "value": 1,
          "unit": "rate",
          "display": "100% (15/15)",
          "n": 15,
          "ci": [
            0.7961,
            1
          ],
          "spanKind": "ci95",
          "context": "GPT-6.1 Sol · effort high · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-pass-rate",
          "series": "Pass rate",
          "point": "GPT-6.1 Sol (medium) · Codex CLI",
          "metric": "h2h-pass-rate",
          "label": "Pass rate on five validated tasks",
          "value": 1,
          "unit": "rate",
          "display": "100% (15/15)",
          "n": 15,
          "ci": [
            0.7961,
            1
          ],
          "spanKind": "ci95",
          "context": "GPT-6.1 Sol · effort medium · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-pass-rate",
          "series": "Pass rate",
          "point": "GPT-6.1 Sol (low) · Codex CLI",
          "metric": "h2h-pass-rate",
          "label": "Pass rate on five validated tasks",
          "value": 1,
          "unit": "rate",
          "display": "100% (10/10)",
          "n": 10,
          "ci": [
            0.7225,
            1
          ],
          "spanKind": "ci95",
          "context": "GPT-6.1 Sol · effort low · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-total-latency",
          "series": "Total time per call",
          "point": "GPT-6.1 Sol (high) · Codex CLI",
          "metric": "h2h-total-latency",
          "label": "Total time per call",
          "value": 5.6,
          "unit": "seconds",
          "display": "5.60 s",
          "n": 15,
          "range": [
            4.05,
            19.52
          ],
          "spanKind": "minmax",
          "context": "GPT-6.1 Sol · effort high · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-total-latency",
          "series": "Total time per call",
          "point": "GPT-6.1 Sol (medium) · Codex CLI",
          "metric": "h2h-total-latency",
          "label": "Total time per call",
          "value": 5.65,
          "unit": "seconds",
          "display": "5.65 s",
          "n": 15,
          "range": [
            4.1,
            25.46
          ],
          "spanKind": "minmax",
          "context": "GPT-6.1 Sol · effort medium · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-total-latency",
          "series": "Total time per call",
          "point": "GPT-6.1 Sol (low) · Codex CLI",
          "metric": "h2h-total-latency",
          "label": "Total time per call",
          "value": 6.26,
          "unit": "seconds",
          "display": "6.26 s",
          "n": 10,
          "range": [
            4.65,
            10.47
          ],
          "spanKind": "minmax",
          "context": "GPT-6.1 Sol · effort low · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-first-useful-latency",
          "series": "Time to first useful output",
          "point": "GPT-6.1 Sol (high) · Codex CLI",
          "metric": "h2h-first-useful-latency",
          "label": "Time to first useful output",
          "value": 5.32,
          "unit": "seconds",
          "display": "5.32 s",
          "n": 15,
          "range": [
            3.64,
            16.37
          ],
          "spanKind": "minmax",
          "context": "GPT-6.1 Sol · effort high · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-first-useful-latency",
          "series": "Time to first useful output",
          "point": "GPT-6.1 Sol (medium) · Codex CLI",
          "metric": "h2h-first-useful-latency",
          "label": "Time to first useful output",
          "value": 5.05,
          "unit": "seconds",
          "display": "5.05 s",
          "n": 15,
          "range": [
            3.36,
            17.82
          ],
          "spanKind": "minmax",
          "context": "GPT-6.1 Sol · effort medium · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-first-useful-latency",
          "series": "Time to first useful output",
          "point": "GPT-6.1 Sol (low) · Codex CLI",
          "metric": "h2h-first-useful-latency",
          "label": "Time to first useful output",
          "value": 5.14,
          "unit": "seconds",
          "display": "5.14 s",
          "n": 10,
          "range": [
            4.02,
            8.5
          ],
          "spanKind": "minmax",
          "context": "GPT-6.1 Sol · effort low · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-input-tokens",
          "series": "Cache read",
          "point": "GPT-6.1 Sol (high) · Codex CLI",
          "metric": "h2h-input-tokens/Cache read",
          "label": "Input tokens per call: what the CLI sends (Cache read)",
          "value": 6716,
          "unit": "tokens",
          "display": "6,716",
          "n": 15,
          "context": "GPT-6.1 Sol · effort high · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-input-tokens",
          "series": "Cache read",
          "point": "GPT-6.1 Sol (medium) · Codex CLI",
          "metric": "h2h-input-tokens/Cache read",
          "label": "Input tokens per call: what the CLI sends (Cache read)",
          "value": 5180,
          "unit": "tokens",
          "display": "5,180",
          "n": 15,
          "context": "GPT-6.1 Sol · effort medium · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-input-tokens",
          "series": "Cache read",
          "point": "GPT-6.1 Sol (low) · Codex CLI",
          "metric": "h2h-input-tokens/Cache read",
          "label": "Input tokens per call: what the CLI sends (Cache read)",
          "value": 8064,
          "unit": "tokens",
          "display": "8,064",
          "n": 10,
          "context": "GPT-6.1 Sol · effort low · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-input-tokens",
          "series": "Other input",
          "point": "GPT-6.1 Sol (high) · Codex CLI",
          "metric": "h2h-input-tokens/Other input",
          "label": "Input tokens per call: what the CLI sends (Other input)",
          "value": 5406,
          "unit": "tokens",
          "display": "5,406",
          "n": 15,
          "context": "GPT-6.1 Sol · effort high · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-input-tokens",
          "series": "Other input",
          "point": "GPT-6.1 Sol (medium) · Codex CLI",
          "metric": "h2h-input-tokens/Other input",
          "label": "Input tokens per call: what the CLI sends (Other input)",
          "value": 6943,
          "unit": "tokens",
          "display": "6,943",
          "n": 15,
          "context": "GPT-6.1 Sol · effort medium · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-input-tokens",
          "series": "Other input",
          "point": "GPT-6.1 Sol (low) · Codex CLI",
          "metric": "h2h-input-tokens/Other input",
          "label": "Input tokens per call: what the CLI sends (Other input)",
          "value": 4059,
          "unit": "tokens",
          "display": "4,059",
          "n": 10,
          "context": "GPT-6.1 Sol · effort low · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-output-tokens",
          "series": "Output tokens",
          "point": "GPT-6.1 Sol (high) · Codex CLI",
          "metric": "h2h-output-tokens/Output tokens",
          "label": "Output tokens per call (Output tokens)",
          "value": 42,
          "unit": "tokens",
          "display": "42",
          "n": 15,
          "context": "GPT-6.1 Sol · effort high · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-output-tokens",
          "series": "Output tokens",
          "point": "GPT-6.1 Sol (medium) · Codex CLI",
          "metric": "h2h-output-tokens/Output tokens",
          "label": "Output tokens per call (Output tokens)",
          "value": 42,
          "unit": "tokens",
          "display": "42",
          "n": 15,
          "context": "GPT-6.1 Sol · effort medium · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-output-tokens",
          "series": "Output tokens",
          "point": "GPT-6.1 Sol (low) · Codex CLI",
          "metric": "h2h-output-tokens/Output tokens",
          "label": "Output tokens per call (Output tokens)",
          "value": 42,
          "unit": "tokens",
          "display": "42",
          "n": 10,
          "context": "GPT-6.1 Sol · effort low · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-list-price-per-call",
          "series": "Cost per call",
          "point": "GPT-6.1 Sol (high) · Codex CLI",
          "metric": "h2h-list-price-per-call",
          "label": "List-price cost per call (calculation)",
          "value": 0.01047,
          "unit": "usd",
          "display": "$0.010",
          "n": 15,
          "range": [
            0.0066,
            0.02812
          ],
          "spanKind": "minmax",
          "calculation": true,
          "context": "GPT-6.1 Sol · effort high · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-list-price-per-call",
          "series": "Cost per call",
          "point": "GPT-6.1 Sol (medium) · Codex CLI",
          "metric": "h2h-list-price-per-call",
          "label": "List-price cost per call (calculation)",
          "value": 0.01018,
          "unit": "usd",
          "display": "$0.010",
          "n": 15,
          "range": [
            0.0054,
            0.02686
          ],
          "spanKind": "minmax",
          "calculation": true,
          "context": "GPT-6.1 Sol · effort medium · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-list-price-per-call",
          "series": "Cost per call",
          "point": "GPT-6.1 Sol (low) · Codex CLI",
          "metric": "h2h-list-price-per-call",
          "label": "List-price cost per call (calculation)",
          "value": 0.00769,
          "unit": "usd",
          "display": "$0.0077",
          "n": 10,
          "range": [
            0.00742,
            0.02649
          ],
          "spanKind": "minmax",
          "calculation": true,
          "context": "GPT-6.1 Sol · effort low · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-cost-per-pass",
          "series": "Cost per pass",
          "point": "GPT-6.1 Sol (low) · Codex CLI",
          "metric": "h2h-cost-per-pass",
          "label": "List-price cost per passing answer (calculation)",
          "value": 0.00998,
          "unit": "usd",
          "display": "$0.010",
          "n": 10,
          "calculation": true,
          "context": "GPT-6.1 Sol · effort low · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-cost-per-pass",
          "series": "Cost per pass",
          "point": "GPT-6.1 Sol (high) · Codex CLI",
          "metric": "h2h-cost-per-pass",
          "label": "List-price cost per passing answer (calculation)",
          "value": 0.01322,
          "unit": "usd",
          "display": "$0.013",
          "n": 15,
          "calculation": true,
          "context": "GPT-6.1 Sol · effort high · five short validated tasks"
        },
        {
          "studySlug": "model-head-to-head",
          "chartId": "h2h-cost-per-pass",
          "series": "Cost per pass",
          "point": "GPT-6.1 Sol (medium) · Codex CLI",
          "metric": "h2h-cost-per-pass",
          "label": "List-price cost per passing answer (calculation)",
          "value": 0.01564,
          "unit": "usd",
          "display": "$0.016",
          "n": 15,
          "calculation": true,
          "context": "GPT-6.1 Sol · effort medium · five short validated tasks"
        },
        {
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-pass-rate",
          "series": "Strict pass",
          "point": "GPT-6.1 Sol (medium) · Codex CLI",
          "metric": "hard-h2h-pass-rate/Strict pass",
          "label": "Pass rate on eight hard tasks (Strict pass)",
          "value": 1,
          "unit": "rate",
          "display": "100% (16/16)",
          "n": 16,
          "ci": [
            0.8064,
            1
          ],
          "spanKind": "ci95",
          "context": "GPT-6.1 Sol · effort medium · eight hard validated tasks"
        },
        {
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-pass-rate",
          "series": "Strict pass",
          "point": "GPT-6.1 Sol (high) · Codex CLI",
          "metric": "hard-h2h-pass-rate/Strict pass",
          "label": "Pass rate on eight hard tasks (Strict pass)",
          "value": 1,
          "unit": "rate",
          "display": "100% (16/16)",
          "n": 16,
          "ci": [
            0.8064,
            1
          ],
          "spanKind": "ci95",
          "context": "GPT-6.1 Sol · effort high · eight hard validated tasks"
        },
        {
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-pass-rate",
          "series": "Lenient (format misses counted)",
          "point": "GPT-6.1 Sol (medium) · Codex CLI",
          "metric": "hard-h2h-pass-rate/Lenient (format misses counted)",
          "label": "Pass rate on eight hard tasks (Lenient (format misses counted))",
          "value": 1,
          "unit": "rate",
          "display": "100% (16/16)",
          "n": 16,
          "ci": [
            0.8064,
            1
          ],
          "spanKind": "ci95",
          "context": "GPT-6.1 Sol · effort medium · eight hard validated tasks"
        },
        {
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-pass-rate",
          "series": "Lenient (format misses counted)",
          "point": "GPT-6.1 Sol (high) · Codex CLI",
          "metric": "hard-h2h-pass-rate/Lenient (format misses counted)",
          "label": "Pass rate on eight hard tasks (Lenient (format misses counted))",
          "value": 1,
          "unit": "rate",
          "display": "100% (16/16)",
          "n": 16,
          "ci": [
            0.8064,
            1
          ],
          "spanKind": "ci95",
          "context": "GPT-6.1 Sol · effort high · eight hard validated tasks"
        },
        {
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-total-latency",
          "series": "Total time per call on hard tasks (separate batches)",
          "point": "GPT-6.1 Sol (medium) · Codex CLI",
          "metric": "hard-h2h-total-latency",
          "label": "Total time per call on hard tasks (separate batches)",
          "value": 13.11,
          "unit": "seconds",
          "display": "13.1 s",
          "n": 16,
          "range": [
            8.54,
            61.6
          ],
          "spanKind": "minmax",
          "context": "GPT-6.1 Sol · effort medium · eight hard validated tasks"
        },
        {
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-total-latency",
          "series": "Total time per call on hard tasks (separate batches)",
          "point": "GPT-6.1 Sol (high) · Codex CLI",
          "metric": "hard-h2h-total-latency",
          "label": "Total time per call on hard tasks (separate batches)",
          "value": 18.12,
          "unit": "seconds",
          "display": "18.1 s",
          "n": 16,
          "range": [
            11.67,
            92.21
          ],
          "spanKind": "minmax",
          "context": "GPT-6.1 Sol · effort high · eight hard validated tasks"
        },
        {
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-first-useful-latency",
          "series": "Time to first useful output on hard tasks",
          "point": "GPT-6.1 Sol (medium) · Codex CLI",
          "metric": "hard-h2h-first-useful-latency",
          "label": "Time to first useful output on hard tasks",
          "value": 10.23,
          "unit": "seconds",
          "display": "10.2 s",
          "n": 16,
          "range": [
            6.09,
            40.41
          ],
          "spanKind": "minmax",
          "context": "GPT-6.1 Sol · effort medium · eight hard validated tasks"
        },
        {
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-first-useful-latency",
          "series": "Time to first useful output on hard tasks",
          "point": "GPT-6.1 Sol (high) · Codex CLI",
          "metric": "hard-h2h-first-useful-latency",
          "label": "Time to first useful output on hard tasks",
          "value": 12.69,
          "unit": "seconds",
          "display": "12.7 s",
          "n": 16,
          "range": [
            8.93,
            75.91
          ],
          "spanKind": "minmax",
          "context": "GPT-6.1 Sol · effort high · eight hard validated tasks"
        },
        {
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-output-tokens",
          "series": "Output tokens",
          "point": "GPT-6.1 Sol (medium) · Codex CLI",
          "metric": "hard-h2h-output-tokens/Output tokens",
          "label": "Output tokens per call on hard tasks (Output tokens)",
          "value": 335,
          "unit": "tokens",
          "display": "335",
          "n": 16,
          "context": "GPT-6.1 Sol · effort medium · eight hard validated tasks"
        },
        {
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-output-tokens",
          "series": "Output tokens",
          "point": "GPT-6.1 Sol (high) · Codex CLI",
          "metric": "hard-h2h-output-tokens/Output tokens",
          "label": "Output tokens per call on hard tasks (Output tokens)",
          "value": 436,
          "unit": "tokens",
          "display": "436",
          "n": 16,
          "context": "GPT-6.1 Sol · effort high · eight hard validated tasks"
        },
        {
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-cost-per-pass",
          "series": "Cost per strict pass",
          "point": "GPT-6.1 Sol (high) · Codex CLI",
          "metric": "hard-h2h-cost-per-pass",
          "label": "List-price cost per strict pass on hard tasks (calculation)",
          "value": 0.01514,
          "unit": "usd",
          "display": "$0.015",
          "n": 16,
          "calculation": true,
          "context": "GPT-6.1 Sol · effort high · eight hard validated tasks"
        },
        {
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-cost-per-pass",
          "series": "Cost per strict pass",
          "point": "GPT-6.1 Sol (medium) · Codex CLI",
          "metric": "hard-h2h-cost-per-pass",
          "label": "List-price cost per strict pass on hard tasks (calculation)",
          "value": 0.02564,
          "unit": "usd",
          "display": "$0.026",
          "n": 16,
          "calculation": true,
          "context": "GPT-6.1 Sol · effort medium · eight hard validated tasks"
        },
        {
          "studySlug": "coding-agents-head-to-head",
          "chartId": "coding-agents-pass-rate",
          "series": "Passed every hidden check",
          "point": "GPT-6.1 Sol (medium, tester’s AGENTS.md) · Codex CLI",
          "metric": "coding-agents-pass-rate",
          "label": "Coding sessions that passed every hidden check",
          "value": 1,
          "unit": "rate",
          "display": "100% (12/12)",
          "n": 12,
          "ci": [
            0.7575,
            1
          ],
          "spanKind": "ci95",
          "context": "GPT-6.1 Sol · effort medium · tester’s AGENTS.md · six small repository tasks with hidden tests"
        },
        {
          "studySlug": "coding-agents-head-to-head",
          "chartId": "coding-agents-wall-time",
          "series": "Wall time per session",
          "point": "GPT-6.1 Sol (medium, tester’s AGENTS.md) · Codex CLI",
          "metric": "coding-agents-wall-time",
          "label": "Time per coding session",
          "value": 113.4,
          "unit": "seconds",
          "display": "113.4 s",
          "n": 12,
          "range": [
            78.5,
            221.9
          ],
          "spanKind": "minmax",
          "context": "GPT-6.1 Sol · effort medium · tester’s AGENTS.md · six small repository tasks with hidden tests"
        },
        {
          "studySlug": "coding-agents-head-to-head",
          "chartId": "coding-agents-tool-calls",
          "series": "Tool calls per session",
          "point": "GPT-6.1 Sol (medium, tester’s AGENTS.md) · Codex CLI",
          "metric": "coding-agents-tool-calls",
          "label": "Tool calls per coding session",
          "value": 12.5,
          "unit": "calls",
          "display": "12.5",
          "n": 12,
          "range": [
            8,
            18
          ],
          "spanKind": "minmax",
          "context": "GPT-6.1 Sol · effort medium · tester’s AGENTS.md · six small repository tasks with hidden tests"
        },
        {
          "studySlug": "coding-agents-head-to-head",
          "chartId": "coding-agents-cost-per-pass",
          "series": "List-price cost per pass",
          "point": "GPT-6.1 Sol (medium, tester’s AGENTS.md) · Codex CLI",
          "metric": "coding-agents-cost-per-pass",
          "label": "List-price cost per passing coding session (calculation)",
          "value": 0.0978,
          "unit": "usd",
          "display": "$0.098",
          "n": 12,
          "calculation": true,
          "context": "GPT-6.1 Sol · effort medium · tester’s AGENTS.md · six small repository tasks with hidden tests"
        },
        {
          "studySlug": "effort-ladder",
          "chartId": "effort-ladder-pass-rate",
          "series": "Strict pass",
          "point": "GPT-6.1 Sol (low) · Codex CLI",
          "metric": "effort-ladder-pass-rate",
          "label": "Strict pass rate by effort on eight hard tasks",
          "value": 1,
          "unit": "rate",
          "display": "100% (16/16)",
          "n": 16,
          "ci": [
            0.8064,
            1
          ],
          "spanKind": "ci95",
          "context": "GPT-6.1 Sol · effort low · eight hard validated tasks, effort ladder"
        },
        {
          "studySlug": "effort-ladder",
          "chartId": "effort-ladder-pass-rate",
          "series": "Strict pass",
          "point": "GPT-6.1 Sol (medium) · Codex CLI",
          "metric": "effort-ladder-pass-rate",
          "label": "Strict pass rate by effort on eight hard tasks",
          "value": 1,
          "unit": "rate",
          "display": "100% (16/16)",
          "n": 16,
          "ci": [
            0.8064,
            1
          ],
          "spanKind": "ci95",
          "context": "GPT-6.1 Sol · effort medium · eight hard validated tasks, effort ladder"
        },
        {
          "studySlug": "effort-ladder",
          "chartId": "effort-ladder-pass-rate",
          "series": "Strict pass",
          "point": "GPT-6.1 Sol (high) · Codex CLI",
          "metric": "effort-ladder-pass-rate",
          "label": "Strict pass rate by effort on eight hard tasks",
          "value": 1,
          "unit": "rate",
          "display": "100% (16/16)",
          "n": 16,
          "ci": [
            0.8064,
            1
          ],
          "spanKind": "ci95",
          "context": "GPT-6.1 Sol · effort high · eight hard validated tasks, effort ladder"
        },
        {
          "studySlug": "effort-ladder",
          "chartId": "effort-ladder-total-latency",
          "series": "Total time per call by effort on hard tasks",
          "point": "GPT-6.1 Sol (low) · Codex CLI",
          "metric": "effort-ladder-total-latency",
          "label": "Total time per call by effort on hard tasks",
          "value": 13.62,
          "unit": "seconds",
          "display": "13.6 s",
          "n": 16,
          "range": [
            7.94,
            44.29
          ],
          "spanKind": "minmax",
          "context": "GPT-6.1 Sol · effort low · eight hard validated tasks, effort ladder"
        },
        {
          "studySlug": "effort-ladder",
          "chartId": "effort-ladder-total-latency",
          "series": "Total time per call by effort on hard tasks",
          "point": "GPT-6.1 Sol (medium) · Codex CLI",
          "metric": "effort-ladder-total-latency",
          "label": "Total time per call by effort on hard tasks",
          "value": 13.11,
          "unit": "seconds",
          "display": "13.1 s",
          "n": 16,
          "range": [
            8.54,
            61.6
          ],
          "spanKind": "minmax",
          "context": "GPT-6.1 Sol · effort medium · eight hard validated tasks, effort ladder"
        },
        {
          "studySlug": "effort-ladder",
          "chartId": "effort-ladder-total-latency",
          "series": "Total time per call by effort on hard tasks",
          "point": "GPT-6.1 Sol (high) · Codex CLI",
          "metric": "effort-ladder-total-latency",
          "label": "Total time per call by effort on hard tasks",
          "value": 18.12,
          "unit": "seconds",
          "display": "18.1 s",
          "n": 16,
          "range": [
            11.67,
            92.21
          ],
          "spanKind": "minmax",
          "context": "GPT-6.1 Sol · effort high · eight hard validated tasks, effort ladder"
        },
        {
          "studySlug": "effort-ladder",
          "chartId": "effort-ladder-output-tokens",
          "series": "Output tokens",
          "point": "GPT-6.1 Sol (low) · Codex CLI",
          "metric": "effort-ladder-output-tokens/Output tokens",
          "label": "Output tokens per call by effort on hard tasks (Output tokens)",
          "value": 284,
          "unit": "tokens",
          "display": "284",
          "n": 16,
          "context": "GPT-6.1 Sol · effort low · eight hard validated tasks, effort ladder"
        },
        {
          "studySlug": "effort-ladder",
          "chartId": "effort-ladder-output-tokens",
          "series": "Output tokens",
          "point": "GPT-6.1 Sol (medium) · Codex CLI",
          "metric": "effort-ladder-output-tokens/Output tokens",
          "label": "Output tokens per call by effort on hard tasks (Output tokens)",
          "value": 335,
          "unit": "tokens",
          "display": "335",
          "n": 16,
          "context": "GPT-6.1 Sol · effort medium · eight hard validated tasks, effort ladder"
        },
        {
          "studySlug": "effort-ladder",
          "chartId": "effort-ladder-output-tokens",
          "series": "Output tokens",
          "point": "GPT-6.1 Sol (high) · Codex CLI",
          "metric": "effort-ladder-output-tokens/Output tokens",
          "label": "Output tokens per call by effort on hard tasks (Output tokens)",
          "value": 436,
          "unit": "tokens",
          "display": "436",
          "n": 16,
          "context": "GPT-6.1 Sol · effort high · eight hard validated tasks, effort ladder"
        },
        {
          "studySlug": "effort-ladder",
          "chartId": "effort-ladder-cost-per-pass",
          "series": "Cost per strict pass",
          "point": "GPT-6.1 Sol (low) · Codex CLI",
          "metric": "effort-ladder-cost-per-pass",
          "label": "List-price cost per strict pass by effort (calculation)",
          "value": 0.01284,
          "unit": "usd",
          "display": "$0.013",
          "n": 16,
          "calculation": true,
          "context": "GPT-6.1 Sol · effort low · eight hard validated tasks, effort ladder"
        },
        {
          "studySlug": "effort-ladder",
          "chartId": "effort-ladder-cost-per-pass",
          "series": "Cost per strict pass",
          "point": "GPT-6.1 Sol (medium) · Codex CLI",
          "metric": "effort-ladder-cost-per-pass",
          "label": "List-price cost per strict pass by effort (calculation)",
          "value": 0.02564,
          "unit": "usd",
          "display": "$0.026",
          "n": 16,
          "calculation": true,
          "context": "GPT-6.1 Sol · effort medium · eight hard validated tasks, effort ladder"
        },
        {
          "studySlug": "effort-ladder",
          "chartId": "effort-ladder-cost-per-pass",
          "series": "Cost per strict pass",
          "point": "GPT-6.1 Sol (high) · Codex CLI",
          "metric": "effort-ladder-cost-per-pass",
          "label": "List-price cost per strict pass by effort (calculation)",
          "value": 0.01514,
          "unit": "usd",
          "display": "$0.015",
          "n": 16,
          "calculation": true,
          "context": "GPT-6.1 Sol · effort high · eight hard validated tasks, effort ladder"
        },
        {
          "studySlug": "caching-consistency",
          "chartId": "consistency-pass-rate",
          "series": "Exact number",
          "point": "GPT-6.1 Sol (medium) · Codex CLI",
          "metric": "consistency-pass-rate/Exact number",
          "label": "Same prompt, 10 times: strict pass rate (Exact number)",
          "value": 1,
          "unit": "rate",
          "display": "100% (10/10)",
          "n": 10,
          "ci": [
            0.7225,
            1
          ],
          "spanKind": "ci95",
          "context": "GPT-6.1 Sol · effort medium · same prompt repeated 10 times"
        },
        {
          "studySlug": "caching-consistency",
          "chartId": "consistency-pass-rate",
          "series": "JSON object",
          "point": "GPT-6.1 Sol (medium) · Codex CLI",
          "metric": "consistency-pass-rate/JSON object",
          "label": "Same prompt, 10 times: strict pass rate (JSON object)",
          "value": 1,
          "unit": "rate",
          "display": "100% (10/10)",
          "n": 10,
          "ci": [
            0.7225,
            1
          ],
          "spanKind": "ci95",
          "context": "GPT-6.1 Sol · effort medium · same prompt repeated 10 times"
        },
        {
          "studySlug": "caching-consistency",
          "chartId": "consistency-pass-rate",
          "series": "Code fix",
          "point": "GPT-6.1 Sol (medium) · Codex CLI",
          "metric": "consistency-pass-rate/Code fix",
          "label": "Same prompt, 10 times: strict pass rate (Code fix)",
          "value": 1,
          "unit": "rate",
          "display": "100% (10/10)",
          "n": 10,
          "ci": [
            0.7225,
            1
          ],
          "spanKind": "ci95",
          "context": "GPT-6.1 Sol · effort medium · same prompt repeated 10 times"
        },
        {
          "studySlug": "caching-consistency",
          "chartId": "consistency-distinct-answers",
          "series": "Exact number",
          "point": "GPT-6.1 Sol (medium) · Codex CLI",
          "metric": "consistency-distinct-answers/Exact number",
          "label": "Same prompt, 10 times: how many different answers (Exact number)",
          "value": 1,
          "unit": "count",
          "display": "1",
          "n": 10,
          "context": "GPT-6.1 Sol · effort medium · same prompt repeated 10 times"
        },
        {
          "studySlug": "caching-consistency",
          "chartId": "consistency-distinct-answers",
          "series": "JSON object",
          "point": "GPT-6.1 Sol (medium) · Codex CLI",
          "metric": "consistency-distinct-answers/JSON object",
          "label": "Same prompt, 10 times: how many different answers (JSON object)",
          "value": 1,
          "unit": "count",
          "display": "1",
          "n": 10,
          "context": "GPT-6.1 Sol · effort medium · same prompt repeated 10 times"
        },
        {
          "studySlug": "caching-consistency",
          "chartId": "consistency-distinct-answers",
          "series": "Code fix",
          "point": "GPT-6.1 Sol (medium) · Codex CLI",
          "metric": "consistency-distinct-answers/Code fix",
          "label": "Same prompt, 10 times: how many different answers (Code fix)",
          "value": 6,
          "unit": "count",
          "display": "6",
          "n": 10,
          "context": "GPT-6.1 Sol · effort medium · same prompt repeated 10 times"
        },
        {
          "studySlug": "caching-consistency",
          "chartId": "consistency-latency-spread",
          "series": "Exact number",
          "point": "GPT-6.1 Sol (medium) · Codex CLI",
          "metric": "consistency-latency-spread/Exact number",
          "label": "Same prompt, 10 times: time per call (Exact number)",
          "value": 13.38,
          "unit": "seconds",
          "display": "13.4 s",
          "n": 10,
          "range": [
            12.29,
            17.97
          ],
          "spanKind": "minmax",
          "context": "GPT-6.1 Sol · effort medium · same prompt repeated 10 times"
        },
        {
          "studySlug": "caching-consistency",
          "chartId": "consistency-latency-spread",
          "series": "JSON object",
          "point": "GPT-6.1 Sol (medium) · Codex CLI",
          "metric": "consistency-latency-spread/JSON object",
          "label": "Same prompt, 10 times: time per call (JSON object)",
          "value": 6.42,
          "unit": "seconds",
          "display": "6.42 s",
          "n": 10,
          "range": [
            5.25,
            8.26
          ],
          "spanKind": "minmax",
          "context": "GPT-6.1 Sol · effort medium · same prompt repeated 10 times"
        },
        {
          "studySlug": "caching-consistency",
          "chartId": "consistency-latency-spread",
          "series": "Code fix",
          "point": "GPT-6.1 Sol (medium) · Codex CLI",
          "metric": "consistency-latency-spread/Code fix",
          "label": "Same prompt, 10 times: time per call (Code fix)",
          "value": 11.29,
          "unit": "seconds",
          "display": "11.3 s",
          "n": 10,
          "range": [
            9.08,
            14.85
          ],
          "spanKind": "minmax",
          "context": "GPT-6.1 Sol · effort medium · same prompt repeated 10 times"
        },
        {
          "studySlug": "routing-overhead",
          "chartId": "cli-startup-tax",
          "series": "First output event",
          "point": "Codex CLI (default model)",
          "metric": "cli-startup-tax/First output event",
          "label": "CLI start-up tax on a one-word answer (First output event)",
          "value": 489,
          "unit": "ms",
          "display": "489 ms",
          "n": 5,
          "range": [
            354,
            1304
          ],
          "spanKind": "minmax",
          "context": "default model · CLI start-up, one-word prompt, 5 runs"
        },
        {
          "studySlug": "routing-overhead",
          "chartId": "cli-startup-tax",
          "series": "First model output",
          "point": "Codex CLI (default model)",
          "metric": "cli-startup-tax/First model output",
          "label": "CLI start-up tax on a one-word answer (First model output)",
          "value": 5059,
          "unit": "ms",
          "display": "5,059 ms",
          "n": 5,
          "range": [
            4391,
            5478
          ],
          "spanKind": "minmax",
          "context": "default model · CLI start-up, one-word prompt, 5 runs"
        },
        {
          "studySlug": "routing-overhead",
          "chartId": "cli-startup-tax",
          "series": "Total wall time",
          "point": "Codex CLI (default model)",
          "metric": "cli-startup-tax/Total wall time",
          "label": "CLI start-up tax on a one-word answer (Total wall time)",
          "value": 5999,
          "unit": "ms",
          "display": "5,999 ms",
          "n": 5,
          "range": [
            5367,
            6506
          ],
          "spanKind": "minmax",
          "context": "default model · CLI start-up, one-word prompt, 5 runs"
        },
        {
          "studySlug": "routing-overhead",
          "chartId": "cli-startup-input-tokens",
          "series": "Input tokens per call",
          "point": "Codex CLI (default model)",
          "metric": "cli-startup-input-tokens",
          "label": "Input tokens a CLI sends for a one-word answer",
          "value": 17051,
          "unit": "tokens",
          "display": "17,051",
          "n": 5,
          "context": "default model · CLI start-up, one-word prompt, 5 runs"
        },
        {
          "studySlug": "cli-model-latency-tokens",
          "chartId": "cli-vs-api-exact-reply-latency",
          "series": "Total time",
          "point": "Codex CLI · GPT-6 Luna · none",
          "metric": "cli-vs-api-exact-reply-latency/Total time",
          "label": "CLI vs API: time for a one-line answer (Total time)",
          "value": 3.19,
          "unit": "seconds",
          "display": "3.19 s",
          "n": 5,
          "range": [
            2.88,
            3.83
          ],
          "spanKind": "minmax",
          "context": "GPT-6 Luna · effort none · fixed exact reply, 5 runs"
        },
        {
          "studySlug": "cli-model-latency-tokens",
          "chartId": "cli-vs-api-exact-reply-latency",
          "series": "Total time",
          "point": "Codex CLI · GPT-6.1 Sol · low",
          "metric": "cli-vs-api-exact-reply-latency/Total time",
          "label": "CLI vs API: time for a one-line answer (Total time)",
          "value": 4.18,
          "unit": "seconds",
          "display": "4.18 s",
          "n": 5,
          "range": [
            3.86,
            4.53
          ],
          "spanKind": "minmax",
          "context": "GPT-6.1 Sol · effort low · fixed exact reply, 5 runs"
        },
        {
          "studySlug": "cli-model-latency-tokens",
          "chartId": "cli-vs-api-exact-reply-latency",
          "series": "Total time",
          "point": "Codex CLI · GPT-6.1 Sol · high",
          "metric": "cli-vs-api-exact-reply-latency/Total time",
          "label": "CLI vs API: time for a one-line answer (Total time)",
          "value": 4.19,
          "unit": "seconds",
          "display": "4.19 s",
          "n": 5,
          "range": [
            3.81,
            4.69
          ],
          "spanKind": "minmax",
          "context": "GPT-6.1 Sol · effort high · fixed exact reply, 5 runs"
        },
        {
          "studySlug": "cli-model-latency-tokens",
          "chartId": "cli-vs-api-exact-reply-latency",
          "series": "First useful output",
          "point": "Codex CLI · GPT-6 Luna · none",
          "metric": "cli-vs-api-exact-reply-latency/First useful output",
          "label": "CLI vs API: time for a one-line answer (First useful output)",
          "value": 2.79,
          "unit": "seconds",
          "display": "2.79 s",
          "n": 5,
          "range": [
            2.46,
            3.42
          ],
          "spanKind": "minmax",
          "context": "GPT-6 Luna · effort none · fixed exact reply, 5 runs"
        },
        {
          "studySlug": "cli-model-latency-tokens",
          "chartId": "cli-vs-api-exact-reply-latency",
          "series": "First useful output",
          "point": "Codex CLI · GPT-6.1 Sol · low",
          "metric": "cli-vs-api-exact-reply-latency/First useful output",
          "label": "CLI vs API: time for a one-line answer (First useful output)",
          "value": 3.75,
          "unit": "seconds",
          "display": "3.75 s",
          "n": 5,
          "range": [
            3.44,
            4.1
          ],
          "spanKind": "minmax",
          "context": "GPT-6.1 Sol · effort low · fixed exact reply, 5 runs"
        },
        {
          "studySlug": "cli-model-latency-tokens",
          "chartId": "cli-vs-api-exact-reply-latency",
          "series": "First useful output",
          "point": "Codex CLI · GPT-6.1 Sol · high",
          "metric": "cli-vs-api-exact-reply-latency/First useful output",
          "label": "CLI vs API: time for a one-line answer (First useful output)",
          "value": 3.79,
          "unit": "seconds",
          "display": "3.79 s",
          "n": 5,
          "range": [
            3.37,
            4.3
          ],
          "spanKind": "minmax",
          "context": "GPT-6.1 Sol · effort high · fixed exact reply, 5 runs"
        },
        {
          "studySlug": "cli-model-latency-tokens",
          "chartId": "cli-vs-api-small-coding-latency",
          "series": "Total time",
          "point": "Codex CLI · GPT-6 Luna · none",
          "metric": "cli-vs-api-small-coding-latency/Total time",
          "label": "CLI vs API: time for a small coding task (Total time)",
          "value": 9.23,
          "unit": "seconds",
          "display": "9.23 s",
          "n": 3,
          "range": [
            8.99,
            11.68
          ],
          "spanKind": "minmax",
          "context": "GPT-6 Luna · effort none · small coding task, 3 runs"
        },
        {
          "studySlug": "cli-model-latency-tokens",
          "chartId": "cli-vs-api-small-coding-latency",
          "series": "Total time",
          "point": "Codex CLI · GPT-6.1 Sol · low",
          "metric": "cli-vs-api-small-coding-latency/Total time",
          "label": "CLI vs API: time for a small coding task (Total time)",
          "value": 14.15,
          "unit": "seconds",
          "display": "14.2 s",
          "n": 3,
          "range": [
            13.02,
            14.41
          ],
          "spanKind": "minmax",
          "context": "GPT-6.1 Sol · effort low · small coding task, 3 runs"
        },
        {
          "studySlug": "cli-model-latency-tokens",
          "chartId": "cli-vs-api-small-coding-latency",
          "series": "Total time",
          "point": "Codex CLI · GPT-6.1 Sol · high",
          "metric": "cli-vs-api-small-coding-latency/Total time",
          "label": "CLI vs API: time for a small coding task (Total time)",
          "value": 17.85,
          "unit": "seconds",
          "display": "17.9 s",
          "n": 3,
          "range": [
            17.68,
            22.42
          ],
          "spanKind": "minmax",
          "context": "GPT-6.1 Sol · effort high · small coding task, 3 runs"
        },
        {
          "studySlug": "cli-model-latency-tokens",
          "chartId": "cli-vs-api-small-coding-latency",
          "series": "First useful output",
          "point": "Codex CLI · GPT-6 Luna · none",
          "metric": "cli-vs-api-small-coding-latency/First useful output",
          "label": "CLI vs API: time for a small coding task (First useful output)",
          "value": 8.68,
          "unit": "seconds",
          "display": "8.68 s",
          "n": 3,
          "range": [
            8.27,
            11.01
          ],
          "spanKind": "minmax",
          "context": "GPT-6 Luna · effort none · small coding task, 3 runs"
        },
        {
          "studySlug": "cli-model-latency-tokens",
          "chartId": "cli-vs-api-small-coding-latency",
          "series": "First useful output",
          "point": "Codex CLI · GPT-6.1 Sol · low",
          "metric": "cli-vs-api-small-coding-latency/First useful output",
          "label": "CLI vs API: time for a small coding task (First useful output)",
          "value": 13.6,
          "unit": "seconds",
          "display": "13.6 s",
          "n": 3,
          "range": [
            12.52,
            13.83
          ],
          "spanKind": "minmax",
          "context": "GPT-6.1 Sol · effort low · small coding task, 3 runs"
        },
        {
          "studySlug": "cli-model-latency-tokens",
          "chartId": "cli-vs-api-small-coding-latency",
          "series": "First useful output",
          "point": "Codex CLI · GPT-6.1 Sol · high",
          "metric": "cli-vs-api-small-coding-latency/First useful output",
          "label": "CLI vs API: time for a small coding task (First useful output)",
          "value": 17.27,
          "unit": "seconds",
          "display": "17.3 s",
          "n": 3,
          "range": [
            17.13,
            21.86
          ],
          "spanKind": "minmax",
          "context": "GPT-6.1 Sol · effort high · small coding task, 3 runs"
        },
        {
          "studySlug": "cli-model-latency-tokens",
          "chartId": "cli-vs-api-prompt-overhead",
          "series": "Input tokens",
          "point": "Codex CLI · GPT-6 Luna · none",
          "metric": "cli-vs-api-prompt-overhead",
          "label": "Hidden prompt: input tokens for the same one-line request",
          "value": 18859,
          "unit": "tokens",
          "display": "18,859",
          "n": 5,
          "context": "GPT-6 Luna · effort none · short fixed tasks"
        },
        {
          "studySlug": "cli-model-latency-tokens",
          "chartId": "cli-vs-api-prompt-overhead",
          "series": "Input tokens",
          "point": "Codex CLI · GPT-6.1 Sol · low",
          "metric": "cli-vs-api-prompt-overhead",
          "label": "Hidden prompt: input tokens for the same one-line request",
          "value": 19551,
          "unit": "tokens",
          "display": "19,551",
          "n": 5,
          "context": "GPT-6.1 Sol · effort low · short fixed tasks"
        },
        {
          "studySlug": "cli-model-latency-tokens",
          "chartId": "cli-vs-api-prompt-overhead",
          "series": "Input tokens",
          "point": "Codex CLI · GPT-6.1 Sol · high",
          "metric": "cli-vs-api-prompt-overhead",
          "label": "Hidden prompt: input tokens for the same one-line request",
          "value": 19555,
          "unit": "tokens",
          "display": "19,555",
          "n": 5,
          "context": "GPT-6.1 Sol · effort high · short fixed tasks"
        },
        {
          "studySlug": "cli-model-latency-tokens",
          "chartId": "scheduler-repair-claude-vs-codex",
          "series": "Total time",
          "point": "Codex CLI · GPT-6.1 Sol · medium",
          "metric": "scheduler-repair-claude-vs-codex/Total time",
          "label": "Repairing a scheduler: Claude Code vs Codex vs API (Total time)",
          "value": 61.16,
          "unit": "seconds",
          "display": "61.2 s",
          "n": 3,
          "range": [
            59.9,
            69.51
          ],
          "spanKind": "minmax",
          "context": "GPT-6.1 Sol · effort medium · scheduler repair, 296 checks, 3 runs"
        },
        {
          "studySlug": "cli-model-latency-tokens",
          "chartId": "scheduler-repair-claude-vs-codex",
          "series": "First useful output",
          "point": "Codex CLI · GPT-6.1 Sol · medium",
          "metric": "scheduler-repair-claude-vs-codex/First useful output",
          "label": "Repairing a scheduler: Claude Code vs Codex vs API (First useful output)",
          "value": 15.56,
          "unit": "seconds",
          "display": "15.6 s",
          "n": 3,
          "range": [
            13.65,
            23.04
          ],
          "spanKind": "minmax",
          "context": "GPT-6.1 Sol · effort medium · scheduler repair, 296 checks, 3 runs"
        },
        {
          "studySlug": "cli-model-latency-tokens",
          "chartId": "scheduler-repair-output-tokens",
          "series": "Output tokens",
          "point": "Codex CLI · GPT-6.1 Sol · medium",
          "metric": "scheduler-repair-output-tokens/Output tokens",
          "label": "Output tokens to repair the scheduler (Output tokens)",
          "value": 1181,
          "unit": "tokens",
          "display": "1,181",
          "n": 3,
          "context": "GPT-6.1 Sol · effort medium · scheduler repair, 296 checks, 3 runs"
        },
        {
          "studySlug": "cli-model-latency-tokens",
          "statId": "exact-reply-cli-over-api",
          "metric": "stat:exact-reply-cli-over-api",
          "label": "Codex CLI vs OpenAI API, median total time for a one-line answer",
          "value": 3.49,
          "unit": "ratio",
          "display": "3.5x slower",
          "n": 30,
          "context": "short fixed tasks"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-pass-rate",
          "series": "Strict pass",
          "point": "GPT-6 Luna (single call) · Codex CLI",
          "metric": "agent-loop-pass-rate",
          "label": "Strict pass rate: single call vs agent loop on eight hard tasks",
          "value": 0.625,
          "unit": "rate",
          "display": "63% (10/16)",
          "n": 16,
          "ci": [
            0.3864,
            0.8152
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "GPT-6 Luna · single call"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-pass-rate",
          "series": "Strict pass",
          "point": "GPT-6 Luna (agent loop) · Codex CLI",
          "metric": "agent-loop-pass-rate",
          "label": "Strict pass rate: single call vs agent loop on eight hard tasks",
          "value": 0.8571,
          "unit": "rate",
          "display": "86% (12/14)",
          "n": 14,
          "ci": [
            0.6006,
            0.9599
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "GPT-6 Luna · agent loop"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "series": "GPT-6 Luna (single call) · Codex CLI",
          "point": "Interval merge fix",
          "metric": "agent-loop-by-task/Interval merge fix",
          "label": "Strict passes per task: single call vs agent loop: Interval merge fix",
          "value": 1,
          "unit": "rate",
          "display": "100% (2/2)",
          "n": 2,
          "ci": [
            0.3424,
            1
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "GPT-6 Luna · single call"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "series": "GPT-6 Luna (single call) · Codex CLI",
          "point": "DST day-length fix",
          "metric": "agent-loop-by-task/DST day-length fix",
          "label": "Strict passes per task: single call vs agent loop: DST day-length fix",
          "value": 1,
          "unit": "rate",
          "display": "100% (2/2)",
          "n": 2,
          "ci": [
            0.3424,
            1
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "GPT-6 Luna · single call"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "series": "GPT-6 Luna (single call) · Codex CLI",
          "point": "CSV parser",
          "metric": "agent-loop-by-task/CSV parser",
          "label": "Strict passes per task: single call vs agent loop: CSV parser",
          "value": 1,
          "unit": "rate",
          "display": "100% (2/2)",
          "n": 2,
          "ci": [
            0.3424,
            1
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "GPT-6 Luna · single call"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "series": "GPT-6 Luna (single call) · Codex CLI",
          "point": "Event-loop order",
          "metric": "agent-loop-by-task/Event-loop order",
          "label": "Strict passes per task: single call vs agent loop: Event-loop order",
          "value": 0,
          "unit": "rate",
          "display": "0% (0/2)",
          "n": 2,
          "ci": [
            0,
            0.6576
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "GPT-6 Luna · single call"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "series": "GPT-6 Luna (single call) · Codex CLI",
          "point": "Room schedule",
          "metric": "agent-loop-by-task/Room schedule",
          "label": "Strict passes per task: single call vs agent loop: Room schedule",
          "value": 0.5,
          "unit": "rate",
          "display": "50% (1/2)",
          "n": 2,
          "ci": [
            0.0945,
            0.9055
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "GPT-6 Luna · single call"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "series": "GPT-6 Luna (single call) · Codex CLI",
          "point": "SemVer regex",
          "metric": "agent-loop-by-task/SemVer regex",
          "label": "Strict passes per task: single call vs agent loop: SemVer regex",
          "value": 0.5,
          "unit": "rate",
          "display": "50% (1/2)",
          "n": 2,
          "ci": [
            0.0945,
            0.9055
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "GPT-6 Luna · single call"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "series": "GPT-6 Luna (single call) · Codex CLI",
          "point": "Money refactor",
          "metric": "agent-loop-by-task/Money refactor",
          "label": "Strict passes per task: single call vs agent loop: Money refactor",
          "value": 0,
          "unit": "rate",
          "display": "0% (0/2)",
          "n": 2,
          "ci": [
            0,
            0.6576
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "GPT-6 Luna · single call"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "series": "GPT-6 Luna (single call) · Codex CLI",
          "point": "SQL report",
          "metric": "agent-loop-by-task/SQL report",
          "label": "Strict passes per task: single call vs agent loop: SQL report",
          "value": 1,
          "unit": "rate",
          "display": "100% (2/2)",
          "n": 2,
          "ci": [
            0.3424,
            1
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "GPT-6 Luna · single call"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "series": "GPT-6 Luna (agent loop) · Codex CLI",
          "point": "Interval merge fix",
          "metric": "agent-loop-by-task/Interval merge fix",
          "label": "Strict passes per task: single call vs agent loop: Interval merge fix",
          "value": 1,
          "unit": "rate",
          "display": "100% (2/2)",
          "n": 2,
          "ci": [
            0.3424,
            1
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "GPT-6 Luna · agent loop"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "series": "GPT-6 Luna (agent loop) · Codex CLI",
          "point": "CSV parser",
          "metric": "agent-loop-by-task/CSV parser",
          "label": "Strict passes per task: single call vs agent loop: CSV parser",
          "value": 0.5,
          "unit": "rate",
          "display": "50% (1/2)",
          "n": 2,
          "ci": [
            0.0945,
            0.9055
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "GPT-6 Luna · agent loop"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "series": "GPT-6 Luna (agent loop) · Codex CLI",
          "point": "Event-loop order",
          "metric": "agent-loop-by-task/Event-loop order",
          "label": "Strict passes per task: single call vs agent loop: Event-loop order",
          "value": 1,
          "unit": "rate",
          "display": "100% (2/2)",
          "n": 2,
          "ci": [
            0.3424,
            1
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "GPT-6 Luna · agent loop"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "series": "GPT-6 Luna (agent loop) · Codex CLI",
          "point": "Room schedule",
          "metric": "agent-loop-by-task/Room schedule",
          "label": "Strict passes per task: single call vs agent loop: Room schedule",
          "value": 1,
          "unit": "rate",
          "display": "100% (2/2)",
          "n": 2,
          "ci": [
            0.3424,
            1
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "GPT-6 Luna · agent loop"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "series": "GPT-6 Luna (agent loop) · Codex CLI",
          "point": "SemVer regex",
          "metric": "agent-loop-by-task/SemVer regex",
          "label": "Strict passes per task: single call vs agent loop: SemVer regex",
          "value": 1,
          "unit": "rate",
          "display": "100% (2/2)",
          "n": 2,
          "ci": [
            0.3424,
            1
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "GPT-6 Luna · agent loop"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "series": "GPT-6 Luna (agent loop) · Codex CLI",
          "point": "Money refactor",
          "metric": "agent-loop-by-task/Money refactor",
          "label": "Strict passes per task: single call vs agent loop: Money refactor",
          "value": 0.5,
          "unit": "rate",
          "display": "50% (1/2)",
          "n": 2,
          "ci": [
            0.0945,
            0.9055
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "GPT-6 Luna · agent loop"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "series": "GPT-6 Luna (agent loop) · Codex CLI",
          "point": "SQL report",
          "metric": "agent-loop-by-task/SQL report",
          "label": "Strict passes per task: single call vs agent loop: SQL report",
          "value": 1,
          "unit": "rate",
          "display": "100% (2/2)",
          "n": 2,
          "ci": [
            0.3424,
            1
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "GPT-6 Luna · agent loop"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-total-time",
          "series": "Total time per attempt",
          "point": "GPT-6 Luna (single call) · Codex CLI",
          "metric": "agent-loop-total-time",
          "label": "Total time per attempt: single call vs agent loop",
          "value": 5.16,
          "unit": "seconds",
          "display": "5.16 s",
          "n": 16,
          "range": [
            3.59,
            11.32
          ],
          "spanKind": "minmax",
          "context": "GPT-6 Luna · single call"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-total-time",
          "series": "Total time per attempt",
          "point": "GPT-6 Luna (agent loop) · Codex CLI",
          "metric": "agent-loop-total-time",
          "label": "Total time per attempt: single call vs agent loop",
          "value": 9.32,
          "unit": "seconds",
          "display": "9.32 s",
          "n": 14,
          "range": [
            3.78,
            15.89
          ],
          "spanKind": "minmax",
          "context": "GPT-6 Luna · agent loop"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-tokens",
          "series": "Input tokens (cache reads included)",
          "point": "GPT-6 Luna (single call) · Codex CLI",
          "metric": "agent-loop-tokens/Input tokens (cache reads included)",
          "label": "Tokens per attempt: single call vs agent loop (Input tokens (cache reads included))",
          "value": 11582,
          "unit": "tokens",
          "display": "11,582",
          "n": 16,
          "range": [
            11526,
            11818
          ],
          "spanKind": "minmax",
          "context": "GPT-6 Luna · single call"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-tokens",
          "series": "Input tokens (cache reads included)",
          "point": "GPT-6 Luna (agent loop) · Codex CLI",
          "metric": "agent-loop-tokens/Input tokens (cache reads included)",
          "label": "Tokens per attempt: single call vs agent loop (Input tokens (cache reads included))",
          "value": 15530,
          "unit": "tokens",
          "display": "15,530",
          "n": 14,
          "range": [
            15391,
            39009
          ],
          "spanKind": "minmax",
          "context": "GPT-6 Luna · agent loop"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-tokens",
          "series": "Output tokens",
          "point": "GPT-6 Luna (single call) · Codex CLI",
          "metric": "agent-loop-tokens/Output tokens",
          "label": "Tokens per attempt: single call vs agent loop (Output tokens)",
          "value": 345,
          "unit": "tokens",
          "display": "345",
          "n": 16,
          "range": [
            36,
            634
          ],
          "spanKind": "minmax",
          "context": "GPT-6 Luna · single call"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-tokens",
          "series": "Output tokens",
          "point": "GPT-6 Luna (agent loop) · Codex CLI",
          "metric": "agent-loop-tokens/Output tokens",
          "label": "Tokens per attempt: single call vs agent loop (Output tokens)",
          "value": 480,
          "unit": "tokens",
          "display": "480",
          "n": 14,
          "range": [
            143,
            858
          ],
          "spanKind": "minmax",
          "context": "GPT-6 Luna · agent loop"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-tool-calls",
          "series": "Tool calls per attempt",
          "point": "GPT-6 Luna (agent loop) · Codex CLI",
          "metric": "agent-loop-tool-calls",
          "label": "Tool calls per agent-loop attempt",
          "value": 0,
          "unit": "count",
          "display": "0",
          "n": 14,
          "range": [
            0,
            1
          ],
          "spanKind": "minmax",
          "context": "GPT-6 Luna · agent loop"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-cost-per-pass",
          "series": "Cost per strict pass",
          "point": "GPT-6 Luna (single call) · Codex CLI",
          "metric": "agent-loop-cost-per-pass",
          "label": "List-price cost per strict pass: single call vs agent loop (calculation)",
          "value": 0.00116,
          "unit": "usd",
          "display": "$0.0012",
          "n": 16,
          "calculation": true,
          "context": "GPT-6 Luna · single call"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-cost-per-pass",
          "series": "Cost per strict pass",
          "point": "GPT-6 Luna (agent loop) · Codex CLI",
          "metric": "agent-loop-cost-per-pass",
          "label": "List-price cost per strict pass: single call vs agent loop (calculation)",
          "value": 0.00099,
          "unit": "usd",
          "display": "$0.00099",
          "n": 14,
          "calculation": true,
          "context": "GPT-6 Luna · agent loop"
        },
        {
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-pass-rate",
          "series": "Strict pass: the whole reply is the right JSON",
          "point": "GPT-6.1 Sol (low, instructions) · Codex CLI",
          "metric": "structured-output-pass-rate/Strict pass: the whole reply is the right JSON",
          "label": "Does a JSON schema raise the pass rate? Instructions vs schema mode (Strict pass: the whole reply is the right JSON)",
          "value": 1,
          "unit": "rate",
          "display": "100% (12/12)",
          "n": 12,
          "ci": [
            0.7575,
            1
          ],
          "spanKind": "ci95",
          "calculation": true,
          "polarity": "higher",
          "context": "GPT-6.1 Sol · effort low · instructions"
        },
        {
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-pass-rate",
          "series": "Strict pass: the whole reply is the right JSON",
          "point": "GPT-6.1 Sol (low, JSON schema) · Codex CLI",
          "metric": "structured-output-pass-rate/Strict pass: the whole reply is the right JSON",
          "label": "Does a JSON schema raise the pass rate? Instructions vs schema mode (Strict pass: the whole reply is the right JSON)",
          "value": 1,
          "unit": "rate",
          "display": "100% (12/12)",
          "n": 12,
          "ci": [
            0.7575,
            1
          ],
          "spanKind": "ci95",
          "calculation": true,
          "polarity": "higher",
          "context": "GPT-6.1 Sol · effort low · JSON schema"
        },
        {
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-pass-rate",
          "series": "Right answer in any format (strict pass or format miss)",
          "point": "GPT-6.1 Sol (low, instructions) · Codex CLI",
          "metric": "structured-output-pass-rate/Right answer in any format (strict pass or format miss)",
          "label": "Does a JSON schema raise the pass rate? Instructions vs schema mode (Right answer in any format (strict pass or format miss))",
          "value": 1,
          "unit": "rate",
          "display": "100% (12/12)",
          "n": 12,
          "ci": [
            0.7575,
            1
          ],
          "spanKind": "ci95",
          "calculation": true,
          "polarity": "higher",
          "context": "GPT-6.1 Sol · effort low · instructions"
        },
        {
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-pass-rate",
          "series": "Right answer in any format (strict pass or format miss)",
          "point": "GPT-6.1 Sol (low, JSON schema) · Codex CLI",
          "metric": "structured-output-pass-rate/Right answer in any format (strict pass or format miss)",
          "label": "Does a JSON schema raise the pass rate? Instructions vs schema mode (Right answer in any format (strict pass or format miss))",
          "value": 1,
          "unit": "rate",
          "display": "100% (12/12)",
          "n": 12,
          "ci": [
            0.7575,
            1
          ],
          "spanKind": "ci95",
          "calculation": true,
          "polarity": "higher",
          "context": "GPT-6.1 Sol · effort low · JSON schema"
        },
        {
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-outcomes",
          "series": "Strict pass",
          "point": "GPT-6.1 Sol (low, instructions) · Codex CLI",
          "metric": "structured-output-outcomes/Strict pass",
          "label": "What each call produced: strict pass, format miss, wrong values or error (Strict pass)",
          "value": 12,
          "unit": "count",
          "display": "12",
          "n": 12,
          "context": "GPT-6.1 Sol · effort low · instructions"
        },
        {
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-outcomes",
          "series": "Strict pass",
          "point": "GPT-6.1 Sol (low, JSON schema) · Codex CLI",
          "metric": "structured-output-outcomes/Strict pass",
          "label": "What each call produced: strict pass, format miss, wrong values or error (Strict pass)",
          "value": 12,
          "unit": "count",
          "display": "12",
          "n": 12,
          "context": "GPT-6.1 Sol · effort low · JSON schema"
        },
        {
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-outcomes",
          "series": "Format miss",
          "point": "GPT-6.1 Sol (low, instructions) · Codex CLI",
          "metric": "structured-output-outcomes/Format miss",
          "label": "What each call produced: strict pass, format miss, wrong values or error (Format miss)",
          "value": 0,
          "unit": "count",
          "display": "0",
          "n": 12,
          "context": "GPT-6.1 Sol · effort low · instructions"
        },
        {
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-outcomes",
          "series": "Format miss",
          "point": "GPT-6.1 Sol (low, JSON schema) · Codex CLI",
          "metric": "structured-output-outcomes/Format miss",
          "label": "What each call produced: strict pass, format miss, wrong values or error (Format miss)",
          "value": 0,
          "unit": "count",
          "display": "0",
          "n": 12,
          "context": "GPT-6.1 Sol · effort low · JSON schema"
        },
        {
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-outcomes",
          "series": "Wrong values",
          "point": "GPT-6.1 Sol (low, instructions) · Codex CLI",
          "metric": "structured-output-outcomes/Wrong values",
          "label": "What each call produced: strict pass, format miss, wrong values or error (Wrong values)",
          "value": 0,
          "unit": "count",
          "display": "0",
          "n": 12,
          "context": "GPT-6.1 Sol · effort low · instructions"
        },
        {
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-outcomes",
          "series": "Wrong values",
          "point": "GPT-6.1 Sol (low, JSON schema) · Codex CLI",
          "metric": "structured-output-outcomes/Wrong values",
          "label": "What each call produced: strict pass, format miss, wrong values or error (Wrong values)",
          "value": 0,
          "unit": "count",
          "display": "0",
          "n": 12,
          "context": "GPT-6.1 Sol · effort low · JSON schema"
        },
        {
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-outcomes",
          "series": "Error",
          "point": "GPT-6.1 Sol (low, instructions) · Codex CLI",
          "metric": "structured-output-outcomes/Error",
          "label": "What each call produced: strict pass, format miss, wrong values or error (Error)",
          "value": 0,
          "unit": "count",
          "display": "0",
          "n": 12,
          "context": "GPT-6.1 Sol · effort low · instructions"
        },
        {
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-outcomes",
          "series": "Error",
          "point": "GPT-6.1 Sol (low, JSON schema) · Codex CLI",
          "metric": "structured-output-outcomes/Error",
          "label": "What each call produced: strict pass, format miss, wrong values or error (Error)",
          "value": 0,
          "unit": "count",
          "display": "0",
          "n": 12,
          "context": "GPT-6.1 Sol · effort low · JSON schema"
        },
        {
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-time",
          "series": "Median time per call (the three prompts pooled)",
          "point": "GPT-6.1 Sol (low, instructions) · Codex CLI",
          "metric": "structured-output-time",
          "label": "Time per call, instructions vs schema mode",
          "value": 6.21,
          "unit": "seconds",
          "display": "6.21 s",
          "n": 12,
          "range": [
            4.2,
            12.27
          ],
          "spanKind": "minmax",
          "context": "GPT-6.1 Sol · effort low · instructions"
        },
        {
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-time",
          "series": "Median time per call (the three prompts pooled)",
          "point": "GPT-6.1 Sol (low, JSON schema) · Codex CLI",
          "metric": "structured-output-time",
          "label": "Time per call, instructions vs schema mode",
          "value": 5.96,
          "unit": "seconds",
          "display": "5.96 s",
          "n": 12,
          "range": [
            4.62,
            20.97
          ],
          "spanKind": "minmax",
          "context": "GPT-6.1 Sol · effort low · JSON schema"
        },
        {
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-tokens",
          "series": "Median output tokens per call",
          "point": "GPT-6.1 Sol (low, instructions) · Codex CLI",
          "metric": "structured-output-tokens/Median output tokens per call",
          "label": "Output and reasoning tokens per call, instructions vs schema mode (Median output tokens per call)",
          "value": 117,
          "unit": "tokens",
          "display": "117",
          "n": 12,
          "context": "GPT-6.1 Sol · effort low · instructions"
        },
        {
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-tokens",
          "series": "Median output tokens per call",
          "point": "GPT-6.1 Sol (low, JSON schema) · Codex CLI",
          "metric": "structured-output-tokens/Median output tokens per call",
          "label": "Output and reasoning tokens per call, instructions vs schema mode (Median output tokens per call)",
          "value": 123,
          "unit": "tokens",
          "display": "123",
          "n": 12,
          "context": "GPT-6.1 Sol · effort low · JSON schema"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-share",
          "series": "Median call",
          "point": "GPT-6.1 Sol (high) · Codex CLI",
          "metric": "thinking-bill-share",
          "label": "Reasoning share of output tokens per call on hard tasks (calculation)",
          "value": 57.01,
          "unit": "percent",
          "display": "57%",
          "n": 16,
          "range": [
            29.19,
            90.8
          ],
          "spanKind": "minmax",
          "calculation": true,
          "polarity": "none",
          "context": "GPT-6.1 Sol · effort high"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-share",
          "series": "Median call",
          "point": "GPT-6.1 Sol (medium) · Codex CLI",
          "metric": "thinking-bill-share",
          "label": "Reasoning share of output tokens per call on hard tasks (calculation)",
          "value": 46.33,
          "unit": "percent",
          "display": "46.3%",
          "n": 16,
          "range": [
            11.42,
            86.85
          ],
          "spanKind": "minmax",
          "calculation": true,
          "polarity": "none",
          "context": "GPT-6.1 Sol · effort medium"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-cost-per-call",
          "series": "Reasoning (output tokens)",
          "point": "GPT-6.1 Sol (medium) · Codex CLI",
          "metric": "thinking-bill-cost-per-call/Reasoning (output tokens)",
          "label": "List-price cost per call: reasoning, remaining output and input (calculation) (Reasoning (output tokens))",
          "value": 0.002273,
          "unit": "usd",
          "display": "$0.0023",
          "n": 16,
          "calculation": true,
          "polarity": "none",
          "context": "GPT-6.1 Sol · effort medium"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-cost-per-call",
          "series": "Reasoning (output tokens)",
          "point": "GPT-6.1 Sol (high) · Codex CLI",
          "metric": "thinking-bill-cost-per-call/Reasoning (output tokens)",
          "label": "List-price cost per call: reasoning, remaining output and input (calculation) (Reasoning (output tokens))",
          "value": 0.004114,
          "unit": "usd",
          "display": "$0.0041",
          "n": 16,
          "calculation": true,
          "polarity": "none",
          "context": "GPT-6.1 Sol · effort high"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-cost-per-call",
          "series": "Remaining output (visible-answer estimate)",
          "point": "GPT-6.1 Sol (medium) · Codex CLI",
          "metric": "thinking-bill-cost-per-call/Remaining output (visible-answer estimate)",
          "label": "List-price cost per call: reasoning, remaining output and input (calculation) (Remaining output (visible-answer estimate))",
          "value": 0.003025,
          "unit": "usd",
          "display": "$0.0030",
          "n": 16,
          "calculation": true,
          "polarity": "none",
          "context": "GPT-6.1 Sol · effort medium"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-cost-per-call",
          "series": "Remaining output (visible-answer estimate)",
          "point": "GPT-6.1 Sol (high) · Codex CLI",
          "metric": "thinking-bill-cost-per-call/Remaining output (visible-answer estimate)",
          "label": "List-price cost per call: reasoning, remaining output and input (calculation) (Remaining output (visible-answer estimate))",
          "value": 0.002905,
          "unit": "usd",
          "display": "$0.0029",
          "n": 16,
          "calculation": true,
          "polarity": "none",
          "context": "GPT-6.1 Sol · effort high"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-cost-per-call",
          "series": "Input (prompt, cache priced)",
          "point": "GPT-6.1 Sol (medium) · Codex CLI",
          "metric": "thinking-bill-cost-per-call/Input (prompt, cache priced)",
          "label": "List-price cost per call: reasoning, remaining output and input (calculation) (Input (prompt, cache priced))",
          "value": 0.020339,
          "unit": "usd",
          "display": "$0.020",
          "n": 16,
          "calculation": true,
          "polarity": "none",
          "context": "GPT-6.1 Sol · effort medium"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-cost-per-call",
          "series": "Input (prompt, cache priced)",
          "point": "GPT-6.1 Sol (high) · Codex CLI",
          "metric": "thinking-bill-cost-per-call/Input (prompt, cache priced)",
          "label": "List-price cost per call: reasoning, remaining output and input (calculation) (Input (prompt, cache priced))",
          "value": 0.008117,
          "unit": "usd",
          "display": "$0.0081",
          "n": 16,
          "calculation": true,
          "polarity": "none",
          "context": "GPT-6.1 Sol · effort high"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-by-effort",
          "series": "Reasoning cost per strict pass",
          "point": "GPT-6.1 Sol (low) · Codex CLI",
          "metric": "thinking-bill-by-effort/Reasoning cost per strict pass",
          "label": "Reasoning cost per strict pass by effort, with the total (calculation) (Reasoning cost per strict pass)",
          "value": 0.001223,
          "unit": "usd",
          "display": "$0.0012",
          "n": 16,
          "calculation": true,
          "polarity": "none",
          "context": "GPT-6.1 Sol · effort low"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-by-effort",
          "series": "Reasoning cost per strict pass",
          "point": "GPT-6.1 Sol (medium) · Codex CLI",
          "metric": "thinking-bill-by-effort/Reasoning cost per strict pass",
          "label": "Reasoning cost per strict pass by effort, with the total (calculation) (Reasoning cost per strict pass)",
          "value": 0.002273,
          "unit": "usd",
          "display": "$0.0023",
          "n": 16,
          "calculation": true,
          "polarity": "none",
          "context": "GPT-6.1 Sol · effort medium"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-by-effort",
          "series": "Reasoning cost per strict pass",
          "point": "GPT-6.1 Sol (high) · Codex CLI",
          "metric": "thinking-bill-by-effort/Reasoning cost per strict pass",
          "label": "Reasoning cost per strict pass by effort, with the total (calculation) (Reasoning cost per strict pass)",
          "value": 0.004114,
          "unit": "usd",
          "display": "$0.0041",
          "n": 16,
          "calculation": true,
          "polarity": "none",
          "context": "GPT-6.1 Sol · effort high"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-by-effort",
          "series": "Total cost per strict pass",
          "point": "GPT-6.1 Sol (low) · Codex CLI",
          "metric": "thinking-bill-by-effort/Total cost per strict pass",
          "label": "Reasoning cost per strict pass by effort, with the total (calculation) (Total cost per strict pass)",
          "value": 0.012837,
          "unit": "usd",
          "display": "$0.013",
          "n": 16,
          "calculation": true,
          "polarity": "none",
          "context": "GPT-6.1 Sol · effort low"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-by-effort",
          "series": "Total cost per strict pass",
          "point": "GPT-6.1 Sol (medium) · Codex CLI",
          "metric": "thinking-bill-by-effort/Total cost per strict pass",
          "label": "Reasoning cost per strict pass by effort, with the total (calculation) (Total cost per strict pass)",
          "value": 0.025637,
          "unit": "usd",
          "display": "$0.026",
          "n": 16,
          "calculation": true,
          "polarity": "none",
          "context": "GPT-6.1 Sol · effort medium"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-by-effort",
          "series": "Total cost per strict pass",
          "point": "GPT-6.1 Sol (high) · Codex CLI",
          "metric": "thinking-bill-by-effort/Total cost per strict pass",
          "label": "Reasoning cost per strict pass by effort, with the total (calculation) (Total cost per strict pass)",
          "value": 0.015137,
          "unit": "usd",
          "display": "$0.015",
          "n": 16,
          "calculation": true,
          "polarity": "none",
          "context": "GPT-6.1 Sol · effort high"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-short-vs-hard",
          "series": "Eight hard tasks",
          "point": "GPT-6.1 Sol (medium) · Codex CLI",
          "metric": "thinking-bill-short-vs-hard/Eight hard tasks",
          "label": "Reasoning share on short tasks vs hard tasks (calculation) (Eight hard tasks)",
          "value": 46.33,
          "unit": "percent",
          "display": "46.3%",
          "n": 16,
          "range": [
            11.42,
            86.85
          ],
          "spanKind": "minmax",
          "calculation": true,
          "polarity": "none",
          "context": "GPT-6.1 Sol · effort medium"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-short-vs-hard",
          "series": "Eight hard tasks",
          "point": "GPT-6.1 Sol (high) · Codex CLI",
          "metric": "thinking-bill-short-vs-hard/Eight hard tasks",
          "label": "Reasoning share on short tasks vs hard tasks (calculation) (Eight hard tasks)",
          "value": 57.01,
          "unit": "percent",
          "display": "57%",
          "n": 16,
          "range": [
            29.19,
            90.8
          ],
          "spanKind": "minmax",
          "calculation": true,
          "polarity": "none",
          "context": "GPT-6.1 Sol · effort high"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-short-vs-hard",
          "series": "Five short tasks",
          "point": "GPT-6.1 Sol (medium) · Codex CLI",
          "metric": "thinking-bill-short-vs-hard/Five short tasks",
          "label": "Reasoning share on short tasks vs hard tasks (calculation) (Five short tasks)",
          "value": 41.05,
          "unit": "percent",
          "display": "41%",
          "n": 15,
          "range": [
            0,
            71.43
          ],
          "spanKind": "minmax",
          "calculation": true,
          "polarity": "none",
          "context": "GPT-6.1 Sol · effort medium"
        },
        {
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-short-vs-hard",
          "series": "Five short tasks",
          "point": "GPT-6.1 Sol (high) · Codex CLI",
          "metric": "thinking-bill-short-vs-hard/Five short tasks",
          "label": "Reasoning share on short tasks vs hard tasks (calculation) (Five short tasks)",
          "value": 58.06,
          "unit": "percent",
          "display": "58.1%",
          "n": 15,
          "range": [
            0,
            75.76
          ],
          "spanKind": "minmax",
          "calculation": true,
          "polarity": "none",
          "context": "GPT-6.1 Sol · effort high"
        },
        {
          "studySlug": "llm-speed-anatomy",
          "chartId": "speed-anatomy-first-text",
          "series": "Time to first text",
          "point": "GPT-6.1 Sol (low) · Codex CLI",
          "metric": "speed-anatomy-first-text",
          "label": "Time to first text: a 250-line answer, six models",
          "value": 3.52,
          "unit": "seconds",
          "display": "3.52 s",
          "n": 4,
          "range": [
            2.75,
            4.42
          ],
          "spanKind": "minmax",
          "context": "GPT-6.1 Sol · effort low"
        },
        {
          "studySlug": "llm-speed-anatomy",
          "chartId": "speed-anatomy-first-text",
          "series": "Time to first text",
          "point": "GPT-6 Luna (low) · Codex CLI",
          "metric": "speed-anatomy-first-text",
          "label": "Time to first text: a 250-line answer, six models",
          "value": 3.3,
          "unit": "seconds",
          "display": "3.30 s",
          "n": 4,
          "range": [
            3.19,
            3.47
          ],
          "spanKind": "minmax",
          "context": "GPT-6 Luna · effort low"
        },
        {
          "studySlug": "llm-speed-anatomy",
          "chartId": "speed-anatomy-output-speed",
          "series": "Visible tokens per second",
          "point": "GPT-6.1 Sol (low) · Codex CLI",
          "metric": "speed-anatomy-output-speed",
          "label": "Output speed after the first text: visible tokens per second (calculation)",
          "value": 79.6,
          "unit": "tokens",
          "display": "80",
          "n": 4,
          "range": [
            71.6,
            80.5
          ],
          "spanKind": "minmax",
          "calculation": true,
          "context": "GPT-6.1 Sol · effort low"
        },
        {
          "studySlug": "llm-speed-anatomy",
          "chartId": "speed-anatomy-output-speed",
          "series": "Visible tokens per second",
          "point": "GPT-6 Luna (low) · Codex CLI",
          "metric": "speed-anatomy-output-speed",
          "label": "Output speed after the first text: visible tokens per second (calculation)",
          "value": 129.1,
          "unit": "tokens",
          "display": "129",
          "n": 4,
          "range": [
            55.5,
            259.1
          ],
          "spanKind": "minmax",
          "calculation": true,
          "context": "GPT-6 Luna · effort low"
        },
        {
          "studySlug": "llm-speed-anatomy",
          "chartId": "speed-anatomy-chars-per-second",
          "series": "Characters per second",
          "point": "GPT-6.1 Sol (low) · Codex CLI",
          "metric": "speed-anatomy-chars-per-second",
          "label": "Output speed in characters per second after the first text (calculation)",
          "value": 323,
          "unit": "count",
          "display": "323",
          "n": 4,
          "range": [
            291,
            327
          ],
          "spanKind": "minmax",
          "calculation": true,
          "context": "GPT-6.1 Sol · effort low"
        },
        {
          "studySlug": "llm-speed-anatomy",
          "chartId": "speed-anatomy-chars-per-second",
          "series": "Characters per second",
          "point": "GPT-6 Luna (low) · Codex CLI",
          "metric": "speed-anatomy-chars-per-second",
          "label": "Output speed in characters per second after the first text (calculation)",
          "value": 524,
          "unit": "count",
          "display": "524",
          "n": 4,
          "range": [
            225,
            1052
          ],
          "spanKind": "minmax",
          "calculation": true,
          "context": "GPT-6 Luna · effort low"
        },
        {
          "studySlug": "llm-speed-anatomy",
          "chartId": "speed-anatomy-prompt-size",
          "series": "GPT-6.1 Sol (low) · Codex CLI",
          "point": "1k",
          "metric": "speed-anatomy-prompt-size/1k",
          "label": "Time to first text as the prompt grows: 1k",
          "value": 3.36,
          "unit": "seconds",
          "display": "3.36 s",
          "n": 3,
          "range": [
            3.36,
            4.75
          ],
          "spanKind": "minmax",
          "calculation": true,
          "context": "GPT-6.1 Sol · effort low"
        },
        {
          "studySlug": "llm-speed-anatomy",
          "chartId": "speed-anatomy-prompt-size",
          "series": "GPT-6.1 Sol (low) · Codex CLI",
          "point": "16k",
          "metric": "speed-anatomy-prompt-size/16k",
          "label": "Time to first text as the prompt grows: 16k",
          "value": 4.02,
          "unit": "seconds",
          "display": "4.02 s",
          "n": 3,
          "range": [
            3.3,
            4.28
          ],
          "spanKind": "minmax",
          "calculation": true,
          "context": "GPT-6.1 Sol · effort low"
        },
        {
          "studySlug": "llm-speed-anatomy",
          "chartId": "speed-anatomy-prompt-size",
          "series": "GPT-6.1 Sol (low) · Codex CLI",
          "point": "64k",
          "metric": "speed-anatomy-prompt-size/64k",
          "label": "Time to first text as the prompt grows: 64k",
          "value": 3.93,
          "unit": "seconds",
          "display": "3.93 s",
          "n": 3,
          "range": [
            3.42,
            4.38
          ],
          "spanKind": "minmax",
          "calculation": true,
          "context": "GPT-6.1 Sol · effort low"
        },
        {
          "studySlug": "llm-speed-anatomy",
          "chartId": "speed-anatomy-total-by-size",
          "series": "1k prompt",
          "point": "GPT-6.1 Sol (low) · Codex CLI",
          "metric": "speed-anatomy-total-by-size/1k prompt",
          "label": "Total time per call by prompt size (1k prompt)",
          "value": 3.43,
          "unit": "seconds",
          "display": "3.43 s",
          "n": 3,
          "range": [
            3.43,
            4.92
          ],
          "spanKind": "minmax",
          "context": "GPT-6.1 Sol · effort low"
        },
        {
          "studySlug": "llm-speed-anatomy",
          "chartId": "speed-anatomy-total-by-size",
          "series": "16k prompt",
          "point": "GPT-6.1 Sol (low) · Codex CLI",
          "metric": "speed-anatomy-total-by-size/16k prompt",
          "label": "Total time per call by prompt size (16k prompt)",
          "value": 4.14,
          "unit": "seconds",
          "display": "4.14 s",
          "n": 3,
          "range": [
            3.96,
            4.68
          ],
          "spanKind": "minmax",
          "context": "GPT-6.1 Sol · effort low"
        },
        {
          "studySlug": "llm-speed-anatomy",
          "chartId": "speed-anatomy-total-by-size",
          "series": "64k prompt",
          "point": "GPT-6.1 Sol (low) · Codex CLI",
          "metric": "speed-anatomy-total-by-size/64k prompt",
          "label": "Total time per call by prompt size (64k prompt)",
          "value": 3.96,
          "unit": "seconds",
          "display": "3.96 s",
          "n": 3,
          "range": [
            3.47,
            4.44
          ],
          "spanKind": "minmax",
          "context": "GPT-6.1 Sol · effort low"
        },
        {
          "studySlug": "llm-speed-anatomy",
          "chartId": "speed-anatomy-lookup-correct",
          "series": "Exact answer",
          "point": "GPT-6.1 Sol (low) · Codex CLI",
          "metric": "speed-anatomy-lookup-correct",
          "label": "Exact lookup answers at the 1k, 16k and 64k prompt-size targets",
          "value": 1,
          "unit": "rate",
          "display": "100% (9/9)",
          "n": 9,
          "ci": [
            0.7009,
            1
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "GPT-6.1 Sol · effort low"
        },
        {
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-pass-rate",
          "series": "Strict pass",
          "point": "GPT-6.1 Sol (medium) · Codex CLI",
          "metric": "harder-h2h-pass-rate/Strict pass",
          "label": "Pass rate on 4 harder tasks (Strict pass)",
          "value": 0.6875,
          "unit": "rate",
          "display": "69% (11/16)",
          "n": 16,
          "ci": [
            0.444,
            0.8584
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "GPT-6.1 Sol · effort medium"
        },
        {
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-pass-rate",
          "series": "Lenient (format misses counted)",
          "point": "GPT-6.1 Sol (medium) · Codex CLI",
          "metric": "harder-h2h-pass-rate/Lenient (format misses counted)",
          "label": "Pass rate on 4 harder tasks (Lenient (format misses counted))",
          "value": 0.6875,
          "unit": "rate",
          "display": "69% (11/16)",
          "n": 16,
          "ci": [
            0.444,
            0.8584
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "GPT-6.1 Sol · effort medium"
        },
        {
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-tool-attempts",
          "series": "Tool attempt",
          "point": "GPT-6.1 Sol (medium) · Codex CLI",
          "metric": "harder-h2h-tool-attempts",
          "label": "Calls that tried a tool although tools were off",
          "value": 0,
          "unit": "rate",
          "display": "0% (0/16)",
          "n": 16,
          "ci": [
            0,
            0.1936
          ],
          "spanKind": "ci95",
          "polarity": "none",
          "context": "GPT-6.1 Sol · effort medium"
        },
        {
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-pass-by-task",
          "series": "GPT-6.1 Sol (medium) · Codex CLI",
          "point": "10x10 nonogram",
          "metric": "harder-h2h-pass-by-task/10x10 nonogram",
          "label": "Strict pass rate by task: 10x10 nonogram",
          "value": 0.75,
          "unit": "rate",
          "display": "75% (3/4)",
          "n": 4,
          "ci": [
            0.3006,
            0.9544
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "GPT-6.1 Sol · effort medium"
        },
        {
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-pass-by-task",
          "series": "GPT-6.1 Sol (medium) · Codex CLI",
          "point": "Sudoku, 22 givens",
          "metric": "harder-h2h-pass-by-task/Sudoku, 22 givens",
          "label": "Strict pass rate by task: Sudoku, 22 givens",
          "value": 0.25,
          "unit": "rate",
          "display": "25% (1/4)",
          "n": 4,
          "ci": [
            0.0456,
            0.6994
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "GPT-6.1 Sol · effort medium"
        },
        {
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-pass-by-task",
          "series": "GPT-6.1 Sol (medium) · Codex CLI",
          "point": "6x6 Skyscrapers",
          "metric": "harder-h2h-pass-by-task/6x6 Skyscrapers",
          "label": "Strict pass rate by task: 6x6 Skyscrapers",
          "value": 1,
          "unit": "rate",
          "display": "100% (4/4)",
          "n": 4,
          "ci": [
            0.5101,
            1
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "GPT-6.1 Sol · effort medium"
        },
        {
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-pass-by-task",
          "series": "GPT-6.1 Sol (medium) · Codex CLI",
          "point": "Seeded shuffle output",
          "metric": "harder-h2h-pass-by-task/Seeded shuffle output",
          "label": "Strict pass rate by task: Seeded shuffle output",
          "value": 0.75,
          "unit": "rate",
          "display": "75% (3/4)",
          "n": 4,
          "ci": [
            0.3006,
            0.9544
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "GPT-6.1 Sol · effort medium"
        },
        {
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-total-latency",
          "series": "Total time per call",
          "point": "GPT-6.1 Sol (medium) · Codex CLI",
          "metric": "harder-h2h-total-latency",
          "label": "Total time per call on harder tasks",
          "value": 120.24,
          "unit": "seconds",
          "display": "120.2 s",
          "n": 13,
          "range": [
            46.24,
            273.46
          ],
          "spanKind": "minmax",
          "context": "GPT-6.1 Sol · effort medium"
        },
        {
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-output-tokens",
          "series": "Output tokens",
          "point": "GPT-6.1 Sol (medium) · Codex CLI",
          "metric": "harder-h2h-output-tokens/Output tokens",
          "label": "Output tokens per call on harder tasks (Output tokens)",
          "value": 4994,
          "unit": "tokens",
          "display": "4,994",
          "n": 13,
          "range": [
            2099,
            13413
          ],
          "spanKind": "minmax",
          "polarity": "none",
          "context": "GPT-6.1 Sol · effort medium"
        },
        {
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-cost-per-pass",
          "series": "Cost per strict pass",
          "point": "GPT-6.1 Sol (medium) · Codex CLI",
          "metric": "harder-h2h-cost-per-pass",
          "label": "List-price cost per strict pass on harder tasks (calculation)",
          "value": 0.08293,
          "unit": "usd",
          "display": "$0.083",
          "n": 16,
          "calculation": true,
          "context": "GPT-6.1 Sol · effort medium"
        }
      ]
    },
    {
      "slug": "jev-1-13",
      "name": "Jev 1.13",
      "vendor": "TypeSafe",
      "kind": "router",
      "description": "A small routing model from TypeSafe that answers typed routing decisions over an API. Priced on input tokens only. Measured here as a router (accuracy, live time per call and cost per 1,000 decisions) and in the System One arena.",
      "aliases": [
        "Jev 1.13"
      ],
      "facts": [
        {
          "studySlug": "system-one-arena",
          "chartId": "arena-accuracy",
          "series": "Accuracy",
          "point": "Jev 1.13",
          "metric": "arena-accuracy",
          "label": "Who decides right? Accuracy on 1,000+ checkable decisions",
          "value": 0.7681,
          "unit": "rate",
          "display": "77% (805/1048)",
          "n": 1048,
          "ci": [
            0.7416,
            0.7927
          ],
          "spanKind": "ci95",
          "context": ""
        },
        {
          "studySlug": "system-one-arena",
          "chartId": "arena-by-suite",
          "series": "Jev 1.13",
          "point": "Games",
          "metric": "arena-by-suite/Games",
          "label": "Where each model is strong: Games",
          "value": 0.4219,
          "unit": "rate",
          "display": "42% (81/192)",
          "n": 192,
          "ci": [
            0.3542,
            0.4926
          ],
          "spanKind": "ci95",
          "context": ""
        },
        {
          "studySlug": "system-one-arena",
          "chartId": "arena-by-suite",
          "series": "Jev 1.13",
          "point": "Logic and thought experiments",
          "metric": "arena-by-suite/Logic and thought experiments",
          "label": "Where each model is strong: Logic and thought experiments",
          "value": 0.7436,
          "unit": "rate",
          "display": "74% (116/156)",
          "n": 156,
          "ci": [
            0.6698,
            0.8057
          ],
          "spanKind": "ci95",
          "context": ""
        },
        {
          "studySlug": "system-one-arena",
          "chartId": "arena-by-suite",
          "series": "Jev 1.13",
          "point": "Policy cases, 20 industries",
          "metric": "arena-by-suite/Policy cases, 20 industries",
          "label": "Where each model is strong: Policy cases, 20 industries",
          "value": 0.8139,
          "unit": "rate",
          "display": "81% (223/274)",
          "n": 274,
          "ci": [
            0.7636,
            0.8555
          ],
          "spanKind": "ci95",
          "context": ""
        },
        {
          "studySlug": "system-one-arena",
          "chartId": "arena-by-suite",
          "series": "Jev 1.13",
          "point": "Usability intents",
          "metric": "arena-by-suite/Usability intents",
          "label": "Where each model is strong: Usability intents",
          "value": 0.9336,
          "unit": "rate",
          "display": "93% (211/226)",
          "n": 226,
          "ci": [
            0.8934,
            0.9594
          ],
          "spanKind": "ci95",
          "context": ""
        },
        {
          "studySlug": "system-one-arena",
          "chartId": "arena-by-suite",
          "series": "Jev 1.13",
          "point": "Stress tests",
          "metric": "arena-by-suite/Stress tests",
          "label": "Where each model is strong: Stress tests",
          "value": 0.87,
          "unit": "rate",
          "display": "87% (174/200)",
          "n": 200,
          "ci": [
            0.8163,
            0.9097
          ],
          "spanKind": "ci95",
          "context": ""
        },
        {
          "studySlug": "system-one-arena",
          "chartId": "arena-latency-same-gpu",
          "series": "Hosted: network included",
          "point": "Jev 1.13",
          "metric": "arena-latency-same-gpu/Hosted: network included",
          "label": "Speed on one GPU: third-party numbers (Hosted: network included)",
          "value": 524.1,
          "unit": "ms",
          "display": "524 ms",
          "range": [
            524.1,
            536
          ],
          "spanKind": "p50-p95",
          "polarity": "lower",
          "context": ""
        },
        {
          "studySlug": "system-one-arena",
          "chartId": "arena-latency-gateway",
          "series": "Median latency, one gateway",
          "point": "Jev 1.13 (TypeSafe)",
          "metric": "arena-latency-gateway",
          "label": "Speed through one gateway: OpenRouter's own numbers",
          "value": 0.17,
          "unit": "seconds",
          "display": "0.17 s",
          "polarity": "lower",
          "context": ""
        },
        {
          "studySlug": "system-one-arena",
          "chartId": "arena-cost-same-provider",
          "series": "USD per 1,000 decisions",
          "point": "Jev 1.13 ($0.042/M, 825 tokens)",
          "metric": "arena-cost-same-provider",
          "label": "Price per 1,000 decisions at one provider's list prices",
          "value": 0.0347,
          "unit": "usd",
          "display": "$0.035",
          "calculation": true,
          "polarity": "lower",
          "context": "$0.042/M · 825 tokens"
        },
        {
          "studySlug": "system-one-arena",
          "chartId": "arena-robust",
          "series": "First presentation",
          "point": "Jev 1.13",
          "metric": "arena-robust/First presentation",
          "label": "Same question, different presentation (First presentation)",
          "value": 0.7681,
          "unit": "rate",
          "display": "77% (805/1048)",
          "n": 1048,
          "ci": [
            0.7416,
            0.7927
          ],
          "spanKind": "ci95",
          "context": ""
        },
        {
          "studySlug": "system-one-arena",
          "chartId": "arena-robust",
          "series": "All three presentations",
          "point": "Jev 1.13",
          "metric": "arena-robust/All three presentations",
          "label": "Same question, different presentation (All three presentations)",
          "value": 0.729,
          "unit": "rate",
          "display": "73% (764/1048)",
          "n": 1048,
          "ci": [
            0.7013,
            0.755
          ],
          "spanKind": "ci95",
          "context": ""
        },
        {
          "studySlug": "system-one-arena",
          "chartId": "arena-flips",
          "series": "Options shuffled",
          "point": "Jev 1.13",
          "metric": "arena-flips/Options shuffled",
          "label": "Decisions that changed when only the presentation changed (Options shuffled)",
          "value": 0.077,
          "unit": "rate",
          "display": "8% (74/961)",
          "n": 961,
          "ci": [
            0.0618,
            0.0956
          ],
          "spanKind": "ci95",
          "polarity": "lower",
          "context": ""
        },
        {
          "studySlug": "system-one-arena",
          "chartId": "arena-flips",
          "series": "Options renamed a, b, c",
          "point": "Jev 1.13",
          "metric": "arena-flips/Options renamed a, b, c",
          "label": "Decisions that changed when only the presentation changed (Options renamed a, b, c)",
          "value": 0.0687,
          "unit": "rate",
          "display": "7% (66/961)",
          "n": 961,
          "ci": [
            0.0543,
            0.0864
          ],
          "spanKind": "ci95",
          "polarity": "lower",
          "context": ""
        },
        {
          "studySlug": "system-one-arena",
          "chartId": "arena-escape",
          "series": "Escaped when it should (higher is better)",
          "point": "Jev 1.13",
          "metric": "arena-escape/Escaped when it should (higher is better)",
          "label": "Knowing when to say \"none of these\" (Escaped when it should (higher is better))",
          "value": 0.8571,
          "unit": "rate",
          "display": "86% (126/147)",
          "n": 147,
          "ci": [
            0.7915,
            0.9046
          ],
          "spanKind": "ci95",
          "polarity": "none",
          "context": ""
        },
        {
          "studySlug": "system-one-arena",
          "chartId": "arena-escape",
          "series": "Escaped when it should not (lower is better)",
          "point": "Jev 1.13",
          "metric": "arena-escape/Escaped when it should not (lower is better)",
          "label": "Knowing when to say \"none of these\" (Escaped when it should not (lower is better))",
          "value": 0.055,
          "unit": "rate",
          "display": "6% (34/618)",
          "n": 618,
          "ci": [
            0.0396,
            0.0759
          ],
          "spanKind": "ci95",
          "polarity": "none",
          "context": ""
        },
        {
          "studySlug": "system-one-arena",
          "chartId": "arena-injection",
          "series": "Right despite the injected text",
          "point": "Jev 1.13",
          "metric": "arena-injection",
          "label": "Prompt injection: does text in the state hijack the decision?",
          "value": 0.95,
          "unit": "rate",
          "display": "95% (38/40)",
          "n": 40,
          "ci": [
            0.835,
            0.9862
          ],
          "spanKind": "ci95",
          "context": ""
        },
        {
          "studySlug": "system-one-arena",
          "chartId": "arena-by-length",
          "series": "Up to 512 tokens",
          "point": "Jev 1.13",
          "metric": "arena-by-length/Up to 512 tokens",
          "label": "Short inputs vs long inputs (Up to 512 tokens)",
          "value": 0.7867,
          "unit": "rate",
          "display": "79% (177/225)",
          "n": 225,
          "ci": [
            0.7286,
            0.8351
          ],
          "spanKind": "ci95",
          "context": ""
        },
        {
          "studySlug": "system-one-arena",
          "chartId": "arena-by-length",
          "series": "More than 512 tokens",
          "point": "Jev 1.13",
          "metric": "arena-by-length/More than 512 tokens",
          "label": "Short inputs vs long inputs (More than 512 tokens)",
          "value": 0.7631,
          "unit": "rate",
          "display": "76% (628/823)",
          "n": 823,
          "ci": [
            0.7328,
            0.7908
          ],
          "spanKind": "ci95",
          "context": ""
        },
        {
          "studySlug": "system-one-arena",
          "chartId": "arena-confident-wrong",
          "series": "Confident wrong answers",
          "point": "Jev 1.13",
          "metric": "arena-confident-wrong",
          "label": "Wrong and sure of it",
          "value": 0.1111,
          "unit": "rate",
          "display": "11% (27/243)",
          "n": 243,
          "ci": [
            0.0775,
            0.1568
          ],
          "spanKind": "ci95",
          "polarity": "lower",
          "context": ""
        },
        {
          "studySlug": "system-one-arena",
          "chartId": "arena-elo",
          "series": "Elo",
          "point": "Jev 1.13",
          "metric": "arena-elo",
          "label": "Tournament rating across every game",
          "value": 1040,
          "unit": "score",
          "display": "1040.00",
          "n": 336,
          "range": [
            1002,
            1082
          ],
          "spanKind": "minmax",
          "context": ""
        },
        {
          "studySlug": "system-one-arena",
          "chartId": "arena-wdl",
          "series": "Wins",
          "point": "Jev 1.13",
          "metric": "arena-wdl/Wins",
          "label": "Wins, draws and losses in the round robin (Wins)",
          "value": 196,
          "unit": "count",
          "display": "196",
          "n": 336,
          "context": ""
        },
        {
          "studySlug": "system-one-arena",
          "chartId": "arena-wdl",
          "series": "Draws",
          "point": "Jev 1.13",
          "metric": "arena-wdl/Draws",
          "label": "Wins, draws and losses in the round robin (Draws)",
          "value": 6,
          "unit": "count",
          "display": "6",
          "n": 336,
          "context": ""
        },
        {
          "studySlug": "system-one-arena",
          "chartId": "arena-wdl",
          "series": "Losses",
          "point": "Jev 1.13",
          "metric": "arena-wdl/Losses",
          "label": "Wins, draws and losses in the round robin (Losses)",
          "value": 134,
          "unit": "count",
          "display": "134",
          "n": 336,
          "context": ""
        },
        {
          "studySlug": "system-one-arena",
          "chartId": "arena-perfect-moves",
          "series": "Jev 1.13",
          "point": "Tic-tac-toe",
          "metric": "arena-perfect-moves/Tic-tac-toe",
          "label": "How often a model found the perfect move: Tic-tac-toe",
          "value": 0.4653,
          "unit": "rate",
          "display": "47% (67/144)",
          "n": 144,
          "ci": [
            0.3858,
            0.5466
          ],
          "spanKind": "ci95",
          "context": ""
        },
        {
          "studySlug": "system-one-arena",
          "chartId": "arena-perfect-moves",
          "series": "Jev 1.13",
          "point": "Connect Four",
          "metric": "arena-perfect-moves/Connect Four",
          "label": "How often a model found the perfect move: Connect Four",
          "value": 0.3789,
          "unit": "rate",
          "display": "38% (133/351)",
          "n": 351,
          "ci": [
            0.3297,
            0.4307
          ],
          "spanKind": "ci95",
          "context": ""
        },
        {
          "studySlug": "system-one-arena",
          "chartId": "arena-perfect-moves",
          "series": "Jev 1.13",
          "point": "Nim",
          "metric": "arena-perfect-moves/Nim",
          "label": "How often a model found the perfect move: Nim",
          "value": 0.3916,
          "unit": "rate",
          "display": "39% (65/166)",
          "n": 166,
          "ci": [
            0.3206,
            0.4675
          ],
          "spanKind": "ci95",
          "context": ""
        },
        {
          "studySlug": "system-one-arena",
          "chartId": "arena-perfect-moves",
          "series": "Jev 1.13",
          "point": "Dots and Boxes",
          "metric": "arena-perfect-moves/Dots and Boxes",
          "label": "How often a model found the perfect move: Dots and Boxes",
          "value": 0.5479,
          "unit": "rate",
          "display": "55% (423/772)",
          "n": 772,
          "ci": [
            0.5127,
            0.5827
          ],
          "spanKind": "ci95",
          "context": ""
        },
        {
          "studySlug": "system-one-arena",
          "chartId": "arena-pong",
          "series": "Pong score",
          "point": "Jev 1.13",
          "metric": "arena-pong",
          "label": "Pong as deployed: who won",
          "value": 0.8125,
          "unit": "rate",
          "display": "81% (13/16)",
          "n": 16,
          "ci": [
            0.5699,
            0.9341
          ],
          "spanKind": "ci95",
          "context": ""
        },
        {
          "studySlug": "system-one-arena",
          "chartId": "arena-pong-quality",
          "series": "Right zone",
          "point": "Jev 1.13",
          "metric": "arena-pong-quality",
          "label": "Pong decision quality, ignoring time",
          "value": 1,
          "unit": "rate",
          "display": "100% (523/523)",
          "n": 523,
          "ci": [
            0.9927,
            1
          ],
          "spanKind": "ci95",
          "context": ""
        },
        {
          "studySlug": "system-one-arena",
          "statId": "arena-jev-latency",
          "metric": "stat:arena-jev-latency",
          "label": "Jev 1.13 hosted API: median call time from Houston, network included (not comparable with a model on another machine)",
          "value": 137,
          "unit": "ms",
          "display": "137 ms",
          "n": 120,
          "context": ""
        },
        {
          "studySlug": "routing-jev-vs-llm",
          "chartId": "routing-exact-decisions",
          "series": "Exact rate",
          "point": "Jev 1.13 (TypeSafe)",
          "metric": "routing-exact-decisions",
          "label": "Typed routing decisions answered exactly right",
          "value": 0.8984,
          "unit": "rate",
          "display": "90%",
          "n": 82,
          "ci": [
            0.8191,
            0.9497
          ],
          "spanKind": "ci95",
          "context": "typed routing decisions · TypeSafe API"
        },
        {
          "studySlug": "routing-jev-vs-llm",
          "chartId": "routing-key-accuracy",
          "series": "Key accuracy",
          "point": "Jev 1.13 (TypeSafe)",
          "metric": "routing-key-accuracy",
          "label": "Per-question accuracy",
          "value": 0.9485,
          "unit": "rate",
          "display": "95% (184/194)",
          "n": 194,
          "ci": [
            0.9077,
            0.9718
          ],
          "spanKind": "ci95",
          "context": "typed routing decisions · TypeSafe API"
        },
        {
          "studySlug": "routing-jev-vs-llm",
          "chartId": "routing-exact-by-decision",
          "series": "Jev 1.13 (TypeSafe)",
          "point": "Failure class",
          "metric": "routing-exact-by-decision/Failure class",
          "label": "Exact rate by decision type: Failure class",
          "value": 1,
          "unit": "rate",
          "display": "100% (18/18)",
          "n": 18,
          "ci": [
            0.8241,
            1
          ],
          "spanKind": "ci95",
          "context": "typed routing decisions · TypeSafe API"
        },
        {
          "studySlug": "routing-jev-vs-llm",
          "chartId": "routing-exact-by-decision",
          "series": "Jev 1.13 (TypeSafe)",
          "point": "Message intent",
          "metric": "routing-exact-by-decision/Message intent",
          "label": "Exact rate by decision type: Message intent",
          "value": 1,
          "unit": "rate",
          "display": "100% (20/20)",
          "n": 20,
          "ci": [
            0.8389,
            1
          ],
          "spanKind": "ci95",
          "context": "typed routing decisions · TypeSafe API"
        },
        {
          "studySlug": "routing-jev-vs-llm",
          "chartId": "routing-exact-by-decision",
          "series": "Jev 1.13 (TypeSafe)",
          "point": "Is it a rule?",
          "metric": "routing-exact-by-decision/Is it a rule?",
          "label": "Exact rate by decision type: Is it a rule?",
          "value": 1,
          "unit": "rate",
          "display": "100% (12/12)",
          "n": 12,
          "ci": [
            0.7575,
            1
          ],
          "spanKind": "ci95",
          "context": "typed routing decisions · TypeSafe API"
        },
        {
          "studySlug": "routing-jev-vs-llm",
          "chartId": "routing-exact-by-decision",
          "series": "Jev 1.13 (TypeSafe)",
          "point": "Context shape",
          "metric": "routing-exact-by-decision/Context shape",
          "label": "Exact rate by decision type: Context shape",
          "value": 0.7396,
          "unit": "rate",
          "display": "74%",
          "n": 32,
          "ci": [
            0.5789,
            0.8675
          ],
          "spanKind": "ci95",
          "context": "typed routing decisions · TypeSafe API"
        },
        {
          "studySlug": "routing-jev-vs-llm",
          "chartId": "routing-cost-per-1000",
          "series": "Cost",
          "point": "Jev 1.13 (TypeSafe)",
          "metric": "routing-cost-per-1000",
          "label": "Cost per 1,000 routing decisions",
          "value": 0.0337,
          "unit": "usd",
          "display": "$0.034",
          "n": 246,
          "calculation": true,
          "context": "typed routing decisions · TypeSafe API"
        },
        {
          "studySlug": "routing-jev-vs-llm",
          "chartId": "routing-decision-latency",
          "series": "Wall time (direct API call)",
          "point": "Jev 1.13 (TypeSafe)",
          "metric": "routing-decision-latency/Wall time (direct API call)",
          "label": "Time per routing decision (Wall time (direct API call))",
          "value": 136.5,
          "unit": "ms",
          "display": "137 ms",
          "n": 246,
          "range": [
            136.5,
            195.7
          ],
          "spanKind": "p50-p95",
          "context": "typed routing decisions · TypeSafe API"
        },
        {
          "studySlug": "routing-overhead",
          "chartId": "router-overhead-decision-latency",
          "series": "Decision time",
          "point": "Jev 1.13 (TypeSafe)",
          "metric": "router-overhead-decision-latency",
          "label": "Time to make one routing decision",
          "value": 136.5,
          "unit": "ms",
          "display": "137 ms",
          "n": 246,
          "range": [
            136.5,
            195.7
          ],
          "spanKind": "p50-p95",
          "context": "routing overhead per decision · TypeSafe API"
        },
        {
          "studySlug": "routing-overhead",
          "chartId": "router-overhead-completed",
          "series": "Completed",
          "point": "Jev 1.13 (TypeSafe)",
          "metric": "router-overhead-completed",
          "label": "Routing calls that returned a decision",
          "value": 1,
          "unit": "rate",
          "display": "100% (246/246)",
          "n": 246,
          "ci": [
            0.9846,
            1
          ],
          "spanKind": "ci95",
          "context": "routing overhead per decision · TypeSafe API"
        },
        {
          "studySlug": "routing-overhead",
          "chartId": "router-overhead-cost-reported",
          "series": "Cost per 1,000 decisions",
          "point": "Jev 1.13 (TypeSafe)",
          "metric": "router-overhead-cost-reported",
          "label": "Cost per 1,000 routing decisions: no model call vs provider-reported",
          "value": 0.0337,
          "unit": "usd",
          "display": "$0.034",
          "n": 82,
          "context": "routing overhead per decision · TypeSafe API"
        },
        {
          "studySlug": "routing-overhead",
          "chartId": "router-overhead-cost-list-price",
          "series": "Cost per 1,000 decisions (list price)",
          "point": "Jev 1.13 (TypeSafe)",
          "metric": "router-overhead-cost-list-price",
          "label": "Cost per 1,000 routing decisions for the model routers (calculation)",
          "value": 0.0337,
          "unit": "usd",
          "display": "$0.034",
          "n": 246,
          "calculation": true,
          "context": "calculation: Agent’s recorded tokens at this model’s list price · routing overhead per decision · TypeSafe API"
        },
        {
          "studySlug": "routing-overhead",
          "chartId": "router-overhead-cost-per-1000-tasks",
          "series": "Every model call routed (49.5 per task)",
          "point": "Jev 1.13 (TypeSafe)",
          "metric": "router-overhead-cost-per-1000-tasks/Every model call routed (49.5 per task)",
          "label": "Added routing cost per 1,000 tasks (calculation) (Every model call routed (49.5 per task))",
          "value": 1.67,
          "unit": "usd",
          "display": "$1.67",
          "calculation": true,
          "context": "calculation per 1,000 tasks from recorded decision counts · TypeSafe API"
        },
        {
          "studySlug": "routing-overhead",
          "chartId": "router-overhead-cost-per-1000-tasks",
          "series": "Only System One decisions (7 per task)",
          "point": "Jev 1.13 (TypeSafe)",
          "metric": "router-overhead-cost-per-1000-tasks/Only System One decisions (7 per task)",
          "label": "Added routing cost per 1,000 tasks (calculation) (Only System One decisions (7 per task))",
          "value": 0.24,
          "unit": "usd",
          "display": "$0.24",
          "calculation": true,
          "context": "calculation per 1,000 tasks from recorded decision counts · TypeSafe API"
        },
        {
          "studySlug": "routing-overhead",
          "chartId": "router-overhead-delay-per-task",
          "series": "Every model call routed (49.5 per task)",
          "point": "Jev 1.13 (TypeSafe)",
          "metric": "router-overhead-delay-per-task/Every model call routed (49.5 per task)",
          "label": "Added routing delay per task (calculation) (Every model call routed (49.5 per task))",
          "value": 6.7568,
          "unit": "seconds",
          "display": "6.76 s",
          "calculation": true,
          "context": "calculation per task from recorded decision counts, decisions in line · TypeSafe API"
        },
        {
          "studySlug": "routing-overhead",
          "chartId": "router-overhead-delay-per-task",
          "series": "Only System One decisions (7 per task)",
          "point": "Jev 1.13 (TypeSafe)",
          "metric": "router-overhead-delay-per-task/Only System One decisions (7 per task)",
          "label": "Added routing delay per task (calculation) (Only System One decisions (7 per task))",
          "value": 0.9555,
          "unit": "seconds",
          "display": "0.96 s",
          "calculation": true,
          "context": "calculation per task from recorded decision counts, decisions in line · TypeSafe API"
        },
        {
          "studySlug": "cost-thought-experiments",
          "chartId": "repriced-cost-per-resolved",
          "series": "Repriced cost per resolved instance",
          "point": "Jev 1.13 (router)",
          "metric": "repriced-cost-per-resolved",
          "label": "Thought experiment: the same tokens at other list prices",
          "value": 0.042,
          "unit": "usd",
          "display": "$0.042",
          "calculation": true,
          "context": "router · calculation: Agent’s recorded tokens at this model’s list price"
        },
        {
          "studySlug": "routing-holdout",
          "chartId": "routing-holdout-exact",
          "series": "Exact rate",
          "point": "Jev 1.13 (TypeSafe)",
          "metric": "routing-holdout-exact",
          "label": "Unseen routing decisions answered exactly right",
          "value": 0.8214,
          "unit": "rate",
          "display": "82% (46/56)",
          "n": 56,
          "ci": [
            0.7016,
            0.9
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": ""
        },
        {
          "studySlug": "routing-holdout",
          "chartId": "routing-holdout-key-accuracy",
          "series": "Key accuracy",
          "point": "Jev 1.13 (TypeSafe)",
          "metric": "routing-holdout-key-accuracy",
          "label": "Per-question accuracy on unseen decisions",
          "value": 0.904,
          "unit": "rate",
          "display": "90% (113/125)",
          "n": 125,
          "ci": [
            0.8397,
            0.9442
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": ""
        },
        {
          "studySlug": "routing-holdout",
          "chartId": "routing-holdout-by-purpose",
          "series": "Jev 1.13 (TypeSafe)",
          "point": "Failure class",
          "metric": "routing-holdout-by-purpose/Failure class",
          "label": "Exact rate on unseen decisions, by decision type: Failure class",
          "value": 0.9286,
          "unit": "rate",
          "display": "93% (13/14)",
          "n": 14,
          "ci": [
            0.6853,
            0.9873
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": ""
        },
        {
          "studySlug": "routing-holdout",
          "chartId": "routing-holdout-by-purpose",
          "series": "Jev 1.13 (TypeSafe)",
          "point": "Message intent",
          "metric": "routing-holdout-by-purpose/Message intent",
          "label": "Exact rate on unseen decisions, by decision type: Message intent",
          "value": 0.8571,
          "unit": "rate",
          "display": "86% (12/14)",
          "n": 14,
          "ci": [
            0.6006,
            0.9599
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": ""
        },
        {
          "studySlug": "routing-holdout",
          "chartId": "routing-holdout-by-purpose",
          "series": "Jev 1.13 (TypeSafe)",
          "point": "Is it a rule?",
          "metric": "routing-holdout-by-purpose/Is it a rule?",
          "label": "Exact rate on unseen decisions, by decision type: Is it a rule?",
          "value": 0.9286,
          "unit": "rate",
          "display": "93% (13/14)",
          "n": 14,
          "ci": [
            0.6853,
            0.9873
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": ""
        },
        {
          "studySlug": "routing-holdout",
          "chartId": "routing-holdout-by-purpose",
          "series": "Jev 1.13 (TypeSafe)",
          "point": "Context shape",
          "metric": "routing-holdout-by-purpose/Context shape",
          "label": "Exact rate on unseen decisions, by decision type: Context shape",
          "value": 0.5714,
          "unit": "rate",
          "display": "57% (8/14)",
          "n": 14,
          "ci": [
            0.3259,
            0.7862
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": ""
        },
        {
          "studySlug": "routing-holdout",
          "chartId": "routing-holdout-tuned-vs-unseen",
          "series": "Tuned set (routing-jev-vs-llm)",
          "point": "Jev 1.13 (TypeSafe)",
          "metric": "routing-holdout-tuned-vs-unseen/Tuned set (routing-jev-vs-llm)",
          "label": "Tuned case set vs unseen holdout: exact rate per router (Tuned set (routing-jev-vs-llm))",
          "value": 0.9024,
          "unit": "rate",
          "display": "90% (74/82)",
          "n": 82,
          "ci": [
            0.8191,
            0.9497
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": ""
        },
        {
          "studySlug": "routing-holdout",
          "chartId": "routing-holdout-tuned-vs-unseen",
          "series": "Unseen holdout",
          "point": "Jev 1.13 (TypeSafe)",
          "metric": "routing-holdout-tuned-vs-unseen/Unseen holdout",
          "label": "Tuned case set vs unseen holdout: exact rate per router (Unseen holdout)",
          "value": 0.8214,
          "unit": "rate",
          "display": "82% (46/56)",
          "n": 56,
          "ci": [
            0.7016,
            0.9
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": ""
        },
        {
          "studySlug": "routing-holdout",
          "chartId": "routing-holdout-latency",
          "series": "Wall time",
          "point": "Jev 1.13 (TypeSafe)",
          "metric": "routing-holdout-latency/Wall time",
          "label": "Time per routing decision, by route (Wall time)",
          "value": 0.139,
          "unit": "seconds",
          "display": "0.14 s",
          "n": 168,
          "range": [
            0.139,
            0.192
          ],
          "spanKind": "p50-p95",
          "context": ""
        },
        {
          "studySlug": "routing-holdout",
          "chartId": "routing-holdout-cost-per-1000",
          "series": "Cost",
          "point": "Jev 1.13 (TypeSafe)",
          "metric": "routing-holdout-cost-per-1000",
          "label": "Cost per 1,000 unseen routing decisions",
          "value": 0.03065,
          "unit": "usd",
          "display": "$0.031",
          "n": 168,
          "calculation": true,
          "context": ""
        },
        {
          "studySlug": "routing-holdout",
          "statId": "holdout-jev-stability",
          "metric": "stat:holdout-jev-stability",
          "label": "Jev 1.13 (TypeSafe): same answers on every key in 3 repetitions",
          "value": 0.9464,
          "unit": "rate",
          "display": "95% (53/56)",
          "n": 56,
          "ci": [
            0.8539,
            0.9816
          ],
          "spanKind": "ci95",
          "context": "TypeSafe"
        },
        {
          "studySlug": "routing-holdout",
          "statId": "holdout-gap-jev",
          "metric": "stat:holdout-gap-jev",
          "label": "Jev 1.13 (TypeSafe): holdout minus tuned-set exact rate",
          "value": -0.081,
          "unit": "rate",
          "display": "−8.1 points",
          "n": 56,
          "calculation": true,
          "context": "TypeSafe"
        }
      ]
    },
    {
      "slug": "agent-harness",
      "name": "Agent",
      "vendor": "Agent",
      "kind": "harness",
      "description": "The Agent coding pipeline: onboarding, research, plan, act, verify and review, on Claude Sonnet 5.5 through a Claude subscription.",
      "aliases": [
        "Agent"
      ],
      "facts": [
        {
          "studySlug": "swe-bench-verified",
          "chartId": "swebench-same-instance-leaderboard",
          "series": "Resolved rate",
          "point": "Agent (Sonnet 5.5, full pipeline)",
          "metric": "swebench-same-instance-leaderboard",
          "label": "Resolved rate on the same 33 SWE-bench Verified instances",
          "value": 0.7576,
          "unit": "rate",
          "display": "76% (25/33)",
          "n": 33,
          "ci": [
            0.5898,
            0.8717
          ],
          "spanKind": "ci95",
          "context": "full pipeline on Claude Sonnet 5.5"
        },
        {
          "studySlug": "swe-bench-verified",
          "chartId": "swebench-by-difficulty-band",
          "series": "Agent",
          "point": "No panel model solved it",
          "metric": "swebench-by-difficulty-band/No panel model solved it",
          "label": "Resolved rate by difficulty band: No panel model solved it",
          "value": 0.25,
          "unit": "rate",
          "display": "25% (1/4)",
          "n": 4,
          "ci": [
            0.0456,
            0.6994
          ],
          "spanKind": "ci95",
          "context": "full pipeline on Claude Sonnet 5.5"
        },
        {
          "studySlug": "swe-bench-verified",
          "chartId": "swebench-by-difficulty-band",
          "series": "Agent",
          "point": "Under half solved it",
          "metric": "swebench-by-difficulty-band/Under half solved it",
          "label": "Resolved rate by difficulty band: Under half solved it",
          "value": 0.75,
          "unit": "rate",
          "display": "75% (3/4)",
          "n": 4,
          "ci": [
            0.3006,
            0.9544
          ],
          "spanKind": "ci95",
          "context": "full pipeline on Claude Sonnet 5.5"
        },
        {
          "studySlug": "swe-bench-verified",
          "chartId": "swebench-by-difficulty-band",
          "series": "Agent",
          "point": "Half or more solved it",
          "metric": "swebench-by-difficulty-band/Half or more solved it",
          "label": "Resolved rate by difficulty band: Half or more solved it",
          "value": 0.8182,
          "unit": "rate",
          "display": "82% (9/11)",
          "n": 11,
          "ci": [
            0.523,
            0.9486
          ],
          "spanKind": "ci95",
          "context": "full pipeline on Claude Sonnet 5.5"
        },
        {
          "studySlug": "swe-bench-verified",
          "chartId": "swebench-by-difficulty-band",
          "series": "Agent",
          "point": "Every panel model solved it",
          "metric": "swebench-by-difficulty-band/Every panel model solved it",
          "label": "Resolved rate by difficulty band: Every panel model solved it",
          "value": 0.8571,
          "unit": "rate",
          "display": "86% (12/14)",
          "n": 14,
          "ci": [
            0.6006,
            0.9599
          ],
          "spanKind": "ci95",
          "context": "full pipeline on Claude Sonnet 5.5"
        },
        {
          "studySlug": "swe-bench-verified",
          "chartId": "swebench-model-calls",
          "series": "Mean calls",
          "point": "Agent",
          "metric": "swebench-model-calls",
          "label": "Model calls per instance",
          "value": 49.5,
          "unit": "calls",
          "display": "49.5",
          "n": 33,
          "context": "full pipeline on Claude Sonnet 5.5"
        },
        {
          "studySlug": "swe-bench-verified",
          "chartId": "swebench-views",
          "series": "Agent",
          "point": "Campaign 1: 25-instance sample",
          "metric": "swebench-views/Campaign 1: 25-instance sample",
          "label": "Every way to slice the run, with intervals: Campaign 1: 25-instance sample",
          "value": 0.72,
          "unit": "rate",
          "display": "72% (18/25)",
          "n": 25,
          "ci": [
            0.5242,
            0.8572
          ],
          "spanKind": "ci95",
          "context": "full pipeline on Claude Sonnet 5.5"
        },
        {
          "studySlug": "swe-bench-verified",
          "chartId": "swebench-views",
          "series": "Agent",
          "point": "Campaign 2: 8 compiled-extension instances",
          "metric": "swebench-views/Campaign 2: 8 compiled-extension instances",
          "label": "Every way to slice the run, with intervals: Campaign 2: 8 compiled-extension instances",
          "value": 0.875,
          "unit": "rate",
          "display": "88% (7/8)",
          "n": 8,
          "ci": [
            0.5291,
            0.9776
          ],
          "spanKind": "ci95",
          "context": "full pipeline on Claude Sonnet 5.5"
        },
        {
          "studySlug": "swe-bench-verified",
          "chartId": "swebench-views",
          "series": "Agent",
          "point": "Original seed draw of 25",
          "metric": "swebench-views/Original seed draw of 25",
          "label": "Every way to slice the run, with intervals: Original seed draw of 25",
          "value": 0.76,
          "unit": "rate",
          "display": "76% (19/25)",
          "n": 25,
          "ci": [
            0.5657,
            0.885
          ],
          "spanKind": "ci95",
          "context": "full pipeline on Claude Sonnet 5.5"
        },
        {
          "studySlug": "swe-bench-verified",
          "chartId": "swebench-views",
          "series": "Agent",
          "point": "All 33 attempted",
          "metric": "swebench-views/All 33 attempted",
          "label": "Every way to slice the run, with intervals: All 33 attempted",
          "value": 0.7576,
          "unit": "rate",
          "display": "76% (25/33)",
          "n": 33,
          "ci": [
            0.5898,
            0.8717
          ],
          "spanKind": "ci95",
          "context": "full pipeline on Claude Sonnet 5.5"
        },
        {
          "studySlug": "swe-bench-verified",
          "statId": "cost-per-attempt",
          "metric": "stat:cost-per-attempt",
          "label": "Agent model cost per attempt (notional)",
          "value": 2.81,
          "unit": "usd",
          "display": "$2.81",
          "n": 33,
          "calculation": true,
          "context": "full pipeline on Claude Sonnet 5.5"
        },
        {
          "studySlug": "swe-bench-verified",
          "statId": "cost-per-resolved",
          "metric": "stat:cost-per-resolved",
          "label": "Agent model cost per resolved instance (notional)",
          "value": 3.71,
          "unit": "usd",
          "display": "$3.71",
          "n": 25,
          "calculation": true,
          "context": "full pipeline on Claude Sonnet 5.5"
        },
        {
          "studySlug": "swe-bench-verified",
          "statId": "median-minutes",
          "metric": "stat:median-minutes",
          "label": "Median worker time per attempt",
          "value": 9.6,
          "unit": "minutes",
          "display": "9.6 min",
          "n": 33,
          "context": "full pipeline on Claude Sonnet 5.5"
        },
        {
          "studySlug": "blind-review-head-to-head",
          "statId": "ai-preferred-latest",
          "metric": "stat:ai-preferred-latest",
          "label": "Tasks where the panel preferred the AI change (latest attempt)",
          "value": 0.75,
          "unit": "rate",
          "display": "75% (9/12)",
          "n": 12,
          "ci": [
            0.4677,
            0.9111
          ],
          "spanKind": "ci95",
          "context": "blind panel: Agent change vs merged human change · blind review panel"
        },
        {
          "studySlug": "blind-review-head-to-head",
          "statId": "ai-preferred-first",
          "metric": "stat:ai-preferred-first",
          "label": "Tasks where the panel preferred the AI change (first scored attempt)",
          "value": 0.5,
          "unit": "rate",
          "display": "50% (6/12)",
          "n": 12,
          "ci": [
            0.2538,
            0.7462
          ],
          "spanKind": "ci95",
          "context": "blind panel: Agent change vs merged human change · blind review panel"
        },
        {
          "studySlug": "blind-review-head-to-head",
          "statId": "ai-preferred-all-pairs",
          "metric": "stat:ai-preferred-all-pairs",
          "label": "All scored pairs where the panel preferred the AI change",
          "value": 0.6,
          "unit": "rate",
          "display": "60% (12/20)",
          "n": 20,
          "ci": [
            0.3866,
            0.7812
          ],
          "spanKind": "ci95",
          "context": "blind panel: Agent change vs merged human change · blind review panel"
        },
        {
          "studySlug": "blind-review-head-to-head",
          "statId": "ai-preferred-public",
          "metric": "stat:ai-preferred-public",
          "label": "Public OSS tasks, latest attempt",
          "value": 0.8,
          "unit": "rate",
          "display": "80% (4/5)",
          "n": 5,
          "ci": [
            0.3755,
            0.9638
          ],
          "spanKind": "ci95",
          "context": "blind panel: Agent change vs merged human change · blind review panel"
        },
        {
          "studySlug": "blind-review-head-to-head",
          "statId": "verdicts-ai",
          "metric": "stat:verdicts-ai",
          "label": "Single critic verdicts that preferred the AI change",
          "value": 0.6894,
          "unit": "rate",
          "display": "69% (91/132)",
          "n": 132,
          "ci": [
            0.606,
            0.762
          ],
          "spanKind": "ci95",
          "context": "blind panel: Agent change vs merged human change · blind review panel"
        },
        {
          "studySlug": "cost-thought-experiments",
          "chartId": "cost-per-resolved-agent-vs-panel",
          "series": "Cost per resolved instance",
          "point": "Agent (notional)",
          "metric": "cost-per-resolved-agent-vs-panel",
          "label": "Recorded cost per resolved instance: Agent vs the public panel",
          "value": 3.706,
          "unit": "usd",
          "display": "$3.71",
          "n": 25,
          "calculation": true,
          "context": "full pipeline on Claude Sonnet 5.5 · notional"
        },
        {
          "studySlug": "coding-calibration",
          "statId": "verified-latest",
          "metric": "stat:verified-latest",
          "label": "Verified deliveries, latest build",
          "value": 1,
          "unit": "count",
          "display": "1 of 3",
          "n": 3,
          "context": "three real tasks, platform builds compared"
        },
        {
          "studySlug": "coding-calibration",
          "statId": "cost-latest",
          "metric": "stat:cost-latest",
          "label": "Notional cost, latest build, all 3 tasks",
          "value": 11.06,
          "unit": "usd",
          "display": "$11.06",
          "n": 3,
          "calculation": true,
          "context": "three real tasks, platform builds compared"
        },
        {
          "studySlug": "coding-calibration",
          "statId": "refusals-trend",
          "metric": "stat:refusals-trend",
          "label": "Guardrail refusals, first vs latest slice",
          "value": 19,
          "unit": "count",
          "display": "26 → 19",
          "n": 3,
          "context": "three real tasks, platform builds compared"
        },
        {
          "studySlug": "coding-calibration",
          "statId": "first-run-cost",
          "metric": "stat:first-run-cost",
          "label": "First calibration run (capped, fastify/session)",
          "value": 4.89,
          "unit": "usd",
          "display": "$4.89, stopped at cap",
          "n": 1,
          "calculation": true,
          "context": "three real tasks, platform builds compared"
        }
      ]
    },
    {
      "slug": "gpt-6-1-sol-openai-api",
      "name": "GPT-6.1 Sol (OpenAI API)",
      "vendor": "OpenAI",
      "kind": "model",
      "description": "OpenAI’s GPT-6.1 Sol model called directly through the OpenAI API, without a CLI.",
      "aliases": [
        "GPT-6.1 Sol · OpenAI API"
      ],
      "facts": [
        {
          "studySlug": "cli-model-latency-tokens",
          "chartId": "cli-vs-api-exact-reply-latency",
          "series": "Total time",
          "point": "OpenAI API · GPT-6.1 Sol · low",
          "metric": "cli-vs-api-exact-reply-latency/Total time",
          "label": "CLI vs API: time for a one-line answer (Total time)",
          "value": 1.02,
          "unit": "seconds",
          "display": "1.02 s",
          "n": 5,
          "range": [
            0.96,
            1.87
          ],
          "spanKind": "minmax",
          "context": "OpenAI API · effort low · fixed exact reply, 5 runs"
        },
        {
          "studySlug": "cli-model-latency-tokens",
          "chartId": "cli-vs-api-exact-reply-latency",
          "series": "Total time",
          "point": "OpenAI API · GPT-6.1 Sol · high",
          "metric": "cli-vs-api-exact-reply-latency/Total time",
          "label": "CLI vs API: time for a one-line answer (Total time)",
          "value": 1.52,
          "unit": "seconds",
          "display": "1.52 s",
          "n": 5,
          "range": [
            1.35,
            2.23
          ],
          "spanKind": "minmax",
          "context": "OpenAI API · effort high · fixed exact reply, 5 runs"
        },
        {
          "studySlug": "cli-model-latency-tokens",
          "chartId": "cli-vs-api-exact-reply-latency",
          "series": "First useful output",
          "point": "OpenAI API · GPT-6.1 Sol · low",
          "metric": "cli-vs-api-exact-reply-latency/First useful output",
          "label": "CLI vs API: time for a one-line answer (First useful output)",
          "value": 0.87,
          "unit": "seconds",
          "display": "0.87 s",
          "n": 5,
          "range": [
            0.84,
            1.74
          ],
          "spanKind": "minmax",
          "context": "OpenAI API · effort low · fixed exact reply, 5 runs"
        },
        {
          "studySlug": "cli-model-latency-tokens",
          "chartId": "cli-vs-api-exact-reply-latency",
          "series": "First useful output",
          "point": "OpenAI API · GPT-6.1 Sol · high",
          "metric": "cli-vs-api-exact-reply-latency/First useful output",
          "label": "CLI vs API: time for a one-line answer (First useful output)",
          "value": 1.34,
          "unit": "seconds",
          "display": "1.34 s",
          "n": 5,
          "range": [
            1.26,
            2.12
          ],
          "spanKind": "minmax",
          "context": "OpenAI API · effort high · fixed exact reply, 5 runs"
        },
        {
          "studySlug": "cli-model-latency-tokens",
          "chartId": "cli-vs-api-small-coding-latency",
          "series": "Total time",
          "point": "OpenAI API · GPT-6.1 Sol · low",
          "metric": "cli-vs-api-small-coding-latency/Total time",
          "label": "CLI vs API: time for a small coding task (Total time)",
          "value": 6,
          "unit": "seconds",
          "display": "6.00 s",
          "n": 3,
          "range": [
            5.44,
            6.2
          ],
          "spanKind": "minmax",
          "context": "OpenAI API · effort low · small coding task, 3 runs"
        },
        {
          "studySlug": "cli-model-latency-tokens",
          "chartId": "cli-vs-api-small-coding-latency",
          "series": "Total time",
          "point": "OpenAI API · GPT-6.1 Sol · high",
          "metric": "cli-vs-api-small-coding-latency/Total time",
          "label": "CLI vs API: time for a small coding task (Total time)",
          "value": 9.56,
          "unit": "seconds",
          "display": "9.56 s",
          "n": 3,
          "range": [
            9.44,
            10.94
          ],
          "spanKind": "minmax",
          "context": "OpenAI API · effort high · small coding task, 3 runs"
        },
        {
          "studySlug": "cli-model-latency-tokens",
          "chartId": "cli-vs-api-small-coding-latency",
          "series": "First useful output",
          "point": "OpenAI API · GPT-6.1 Sol · low",
          "metric": "cli-vs-api-small-coding-latency/First useful output",
          "label": "CLI vs API: time for a small coding task (First useful output)",
          "value": 1.05,
          "unit": "seconds",
          "display": "1.05 s",
          "n": 3,
          "range": [
            0.97,
            1.4
          ],
          "spanKind": "minmax",
          "context": "OpenAI API · effort low · small coding task, 3 runs"
        },
        {
          "studySlug": "cli-model-latency-tokens",
          "chartId": "cli-vs-api-small-coding-latency",
          "series": "First useful output",
          "point": "OpenAI API · GPT-6.1 Sol · high",
          "metric": "cli-vs-api-small-coding-latency/First useful output",
          "label": "CLI vs API: time for a small coding task (First useful output)",
          "value": 5.31,
          "unit": "seconds",
          "display": "5.31 s",
          "n": 3,
          "range": [
            4.99,
            6.42
          ],
          "spanKind": "minmax",
          "context": "OpenAI API · effort high · small coding task, 3 runs"
        },
        {
          "studySlug": "cli-model-latency-tokens",
          "chartId": "cli-vs-api-prompt-overhead",
          "series": "Input tokens",
          "point": "OpenAI API · GPT-6.1 Sol · low",
          "metric": "cli-vs-api-prompt-overhead",
          "label": "Hidden prompt: input tokens for the same one-line request",
          "value": 17,
          "unit": "tokens",
          "display": "17",
          "n": 5,
          "context": "OpenAI API · effort low · short fixed tasks"
        },
        {
          "studySlug": "cli-model-latency-tokens",
          "chartId": "cli-vs-api-prompt-overhead",
          "series": "Input tokens",
          "point": "OpenAI API · GPT-6.1 Sol · high",
          "metric": "cli-vs-api-prompt-overhead",
          "label": "Hidden prompt: input tokens for the same one-line request",
          "value": 17,
          "unit": "tokens",
          "display": "17",
          "n": 5,
          "context": "OpenAI API · effort high · short fixed tasks"
        },
        {
          "studySlug": "cli-model-latency-tokens",
          "chartId": "scheduler-repair-claude-vs-codex",
          "series": "Total time",
          "point": "OpenAI API · GPT-6.1 Sol · medium",
          "metric": "scheduler-repair-claude-vs-codex/Total time",
          "label": "Repairing a scheduler: Claude Code vs Codex vs API (Total time)",
          "value": 17.32,
          "unit": "seconds",
          "display": "17.3 s",
          "n": 3,
          "range": [
            16.28,
            18.61
          ],
          "spanKind": "minmax",
          "context": "OpenAI API · effort medium · scheduler repair, 296 checks, 3 runs"
        },
        {
          "studySlug": "cli-model-latency-tokens",
          "chartId": "scheduler-repair-claude-vs-codex",
          "series": "First useful output",
          "point": "OpenAI API · GPT-6.1 Sol · medium",
          "metric": "scheduler-repair-claude-vs-codex/First useful output",
          "label": "Repairing a scheduler: Claude Code vs Codex vs API (First useful output)",
          "value": 7.46,
          "unit": "seconds",
          "display": "7.46 s",
          "n": 3,
          "range": [
            6.68,
            9.05
          ],
          "spanKind": "minmax",
          "context": "OpenAI API · effort medium · scheduler repair, 296 checks, 3 runs"
        },
        {
          "studySlug": "cli-model-latency-tokens",
          "chartId": "scheduler-repair-output-tokens",
          "series": "Output tokens",
          "point": "OpenAI API · GPT-6.1 Sol · medium",
          "metric": "scheduler-repair-output-tokens/Output tokens",
          "label": "Output tokens to repair the scheduler (Output tokens)",
          "value": 1313,
          "unit": "tokens",
          "display": "1,313",
          "n": 3,
          "context": "OpenAI API · effort medium · scheduler repair, 296 checks, 3 runs"
        }
      ]
    },
    {
      "slug": "gpt-6-luna-codex-cli",
      "name": "GPT-6 Luna (Codex CLI)",
      "vendor": "OpenAI",
      "kind": "model",
      "description": "OpenAI’s GPT-6 Luna model run through the Codex CLI.",
      "aliases": [
        "GPT-6 Luna · Codex CLI"
      ],
      "facts": [
        {
          "studySlug": "cli-model-latency-tokens",
          "chartId": "cli-vs-api-exact-reply-latency",
          "series": "Total time",
          "point": "Codex CLI · GPT-6 Luna · none",
          "metric": "cli-vs-api-exact-reply-latency/Total time",
          "label": "CLI vs API: time for a one-line answer (Total time)",
          "value": 3.19,
          "unit": "seconds",
          "display": "3.19 s",
          "n": 5,
          "range": [
            2.88,
            3.83
          ],
          "spanKind": "minmax",
          "context": "Codex CLI · effort none · fixed exact reply, 5 runs"
        },
        {
          "studySlug": "cli-model-latency-tokens",
          "chartId": "cli-vs-api-exact-reply-latency",
          "series": "First useful output",
          "point": "Codex CLI · GPT-6 Luna · none",
          "metric": "cli-vs-api-exact-reply-latency/First useful output",
          "label": "CLI vs API: time for a one-line answer (First useful output)",
          "value": 2.79,
          "unit": "seconds",
          "display": "2.79 s",
          "n": 5,
          "range": [
            2.46,
            3.42
          ],
          "spanKind": "minmax",
          "context": "Codex CLI · effort none · fixed exact reply, 5 runs"
        },
        {
          "studySlug": "cli-model-latency-tokens",
          "chartId": "cli-vs-api-small-coding-latency",
          "series": "Total time",
          "point": "Codex CLI · GPT-6 Luna · none",
          "metric": "cli-vs-api-small-coding-latency/Total time",
          "label": "CLI vs API: time for a small coding task (Total time)",
          "value": 9.23,
          "unit": "seconds",
          "display": "9.23 s",
          "n": 3,
          "range": [
            8.99,
            11.68
          ],
          "spanKind": "minmax",
          "context": "Codex CLI · effort none · small coding task, 3 runs"
        },
        {
          "studySlug": "cli-model-latency-tokens",
          "chartId": "cli-vs-api-small-coding-latency",
          "series": "First useful output",
          "point": "Codex CLI · GPT-6 Luna · none",
          "metric": "cli-vs-api-small-coding-latency/First useful output",
          "label": "CLI vs API: time for a small coding task (First useful output)",
          "value": 8.68,
          "unit": "seconds",
          "display": "8.68 s",
          "n": 3,
          "range": [
            8.27,
            11.01
          ],
          "spanKind": "minmax",
          "context": "Codex CLI · effort none · small coding task, 3 runs"
        },
        {
          "studySlug": "cli-model-latency-tokens",
          "chartId": "cli-vs-api-prompt-overhead",
          "series": "Input tokens",
          "point": "Codex CLI · GPT-6 Luna · none",
          "metric": "cli-vs-api-prompt-overhead",
          "label": "Hidden prompt: input tokens for the same one-line request",
          "value": 18859,
          "unit": "tokens",
          "display": "18,859",
          "n": 5,
          "context": "Codex CLI · effort none · short fixed tasks"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-pass-rate",
          "series": "Strict pass",
          "point": "GPT-6 Luna (single call) · Codex CLI",
          "metric": "agent-loop-pass-rate",
          "label": "Strict pass rate: single call vs agent loop on eight hard tasks",
          "value": 0.625,
          "unit": "rate",
          "display": "63% (10/16)",
          "n": 16,
          "ci": [
            0.3864,
            0.8152
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Codex CLI · single call"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-pass-rate",
          "series": "Strict pass",
          "point": "GPT-6 Luna (agent loop) · Codex CLI",
          "metric": "agent-loop-pass-rate",
          "label": "Strict pass rate: single call vs agent loop on eight hard tasks",
          "value": 0.8571,
          "unit": "rate",
          "display": "86% (12/14)",
          "n": 14,
          "ci": [
            0.6006,
            0.9599
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Codex CLI · agent loop"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "series": "GPT-6 Luna (single call) · Codex CLI",
          "point": "Interval merge fix",
          "metric": "agent-loop-by-task/Interval merge fix",
          "label": "Strict passes per task: single call vs agent loop: Interval merge fix",
          "value": 1,
          "unit": "rate",
          "display": "100% (2/2)",
          "n": 2,
          "ci": [
            0.3424,
            1
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Codex CLI · single call"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "series": "GPT-6 Luna (single call) · Codex CLI",
          "point": "DST day-length fix",
          "metric": "agent-loop-by-task/DST day-length fix",
          "label": "Strict passes per task: single call vs agent loop: DST day-length fix",
          "value": 1,
          "unit": "rate",
          "display": "100% (2/2)",
          "n": 2,
          "ci": [
            0.3424,
            1
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Codex CLI · single call"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "series": "GPT-6 Luna (single call) · Codex CLI",
          "point": "CSV parser",
          "metric": "agent-loop-by-task/CSV parser",
          "label": "Strict passes per task: single call vs agent loop: CSV parser",
          "value": 1,
          "unit": "rate",
          "display": "100% (2/2)",
          "n": 2,
          "ci": [
            0.3424,
            1
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Codex CLI · single call"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "series": "GPT-6 Luna (single call) · Codex CLI",
          "point": "Event-loop order",
          "metric": "agent-loop-by-task/Event-loop order",
          "label": "Strict passes per task: single call vs agent loop: Event-loop order",
          "value": 0,
          "unit": "rate",
          "display": "0% (0/2)",
          "n": 2,
          "ci": [
            0,
            0.6576
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Codex CLI · single call"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "series": "GPT-6 Luna (single call) · Codex CLI",
          "point": "Room schedule",
          "metric": "agent-loop-by-task/Room schedule",
          "label": "Strict passes per task: single call vs agent loop: Room schedule",
          "value": 0.5,
          "unit": "rate",
          "display": "50% (1/2)",
          "n": 2,
          "ci": [
            0.0945,
            0.9055
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Codex CLI · single call"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "series": "GPT-6 Luna (single call) · Codex CLI",
          "point": "SemVer regex",
          "metric": "agent-loop-by-task/SemVer regex",
          "label": "Strict passes per task: single call vs agent loop: SemVer regex",
          "value": 0.5,
          "unit": "rate",
          "display": "50% (1/2)",
          "n": 2,
          "ci": [
            0.0945,
            0.9055
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Codex CLI · single call"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "series": "GPT-6 Luna (single call) · Codex CLI",
          "point": "Money refactor",
          "metric": "agent-loop-by-task/Money refactor",
          "label": "Strict passes per task: single call vs agent loop: Money refactor",
          "value": 0,
          "unit": "rate",
          "display": "0% (0/2)",
          "n": 2,
          "ci": [
            0,
            0.6576
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Codex CLI · single call"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "series": "GPT-6 Luna (single call) · Codex CLI",
          "point": "SQL report",
          "metric": "agent-loop-by-task/SQL report",
          "label": "Strict passes per task: single call vs agent loop: SQL report",
          "value": 1,
          "unit": "rate",
          "display": "100% (2/2)",
          "n": 2,
          "ci": [
            0.3424,
            1
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Codex CLI · single call"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "series": "GPT-6 Luna (agent loop) · Codex CLI",
          "point": "Interval merge fix",
          "metric": "agent-loop-by-task/Interval merge fix",
          "label": "Strict passes per task: single call vs agent loop: Interval merge fix",
          "value": 1,
          "unit": "rate",
          "display": "100% (2/2)",
          "n": 2,
          "ci": [
            0.3424,
            1
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Codex CLI · agent loop"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "series": "GPT-6 Luna (agent loop) · Codex CLI",
          "point": "CSV parser",
          "metric": "agent-loop-by-task/CSV parser",
          "label": "Strict passes per task: single call vs agent loop: CSV parser",
          "value": 0.5,
          "unit": "rate",
          "display": "50% (1/2)",
          "n": 2,
          "ci": [
            0.0945,
            0.9055
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Codex CLI · agent loop"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "series": "GPT-6 Luna (agent loop) · Codex CLI",
          "point": "Event-loop order",
          "metric": "agent-loop-by-task/Event-loop order",
          "label": "Strict passes per task: single call vs agent loop: Event-loop order",
          "value": 1,
          "unit": "rate",
          "display": "100% (2/2)",
          "n": 2,
          "ci": [
            0.3424,
            1
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Codex CLI · agent loop"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "series": "GPT-6 Luna (agent loop) · Codex CLI",
          "point": "Room schedule",
          "metric": "agent-loop-by-task/Room schedule",
          "label": "Strict passes per task: single call vs agent loop: Room schedule",
          "value": 1,
          "unit": "rate",
          "display": "100% (2/2)",
          "n": 2,
          "ci": [
            0.3424,
            1
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Codex CLI · agent loop"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "series": "GPT-6 Luna (agent loop) · Codex CLI",
          "point": "SemVer regex",
          "metric": "agent-loop-by-task/SemVer regex",
          "label": "Strict passes per task: single call vs agent loop: SemVer regex",
          "value": 1,
          "unit": "rate",
          "display": "100% (2/2)",
          "n": 2,
          "ci": [
            0.3424,
            1
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Codex CLI · agent loop"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "series": "GPT-6 Luna (agent loop) · Codex CLI",
          "point": "Money refactor",
          "metric": "agent-loop-by-task/Money refactor",
          "label": "Strict passes per task: single call vs agent loop: Money refactor",
          "value": 0.5,
          "unit": "rate",
          "display": "50% (1/2)",
          "n": 2,
          "ci": [
            0.0945,
            0.9055
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Codex CLI · agent loop"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "series": "GPT-6 Luna (agent loop) · Codex CLI",
          "point": "SQL report",
          "metric": "agent-loop-by-task/SQL report",
          "label": "Strict passes per task: single call vs agent loop: SQL report",
          "value": 1,
          "unit": "rate",
          "display": "100% (2/2)",
          "n": 2,
          "ci": [
            0.3424,
            1
          ],
          "spanKind": "ci95",
          "polarity": "higher",
          "context": "Codex CLI · agent loop"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-total-time",
          "series": "Total time per attempt",
          "point": "GPT-6 Luna (single call) · Codex CLI",
          "metric": "agent-loop-total-time",
          "label": "Total time per attempt: single call vs agent loop",
          "value": 5.16,
          "unit": "seconds",
          "display": "5.16 s",
          "n": 16,
          "range": [
            3.59,
            11.32
          ],
          "spanKind": "minmax",
          "context": "Codex CLI · single call"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-total-time",
          "series": "Total time per attempt",
          "point": "GPT-6 Luna (agent loop) · Codex CLI",
          "metric": "agent-loop-total-time",
          "label": "Total time per attempt: single call vs agent loop",
          "value": 9.32,
          "unit": "seconds",
          "display": "9.32 s",
          "n": 14,
          "range": [
            3.78,
            15.89
          ],
          "spanKind": "minmax",
          "context": "Codex CLI · agent loop"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-tokens",
          "series": "Input tokens (cache reads included)",
          "point": "GPT-6 Luna (single call) · Codex CLI",
          "metric": "agent-loop-tokens/Input tokens (cache reads included)",
          "label": "Tokens per attempt: single call vs agent loop (Input tokens (cache reads included))",
          "value": 11582,
          "unit": "tokens",
          "display": "11,582",
          "n": 16,
          "range": [
            11526,
            11818
          ],
          "spanKind": "minmax",
          "context": "Codex CLI · single call"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-tokens",
          "series": "Input tokens (cache reads included)",
          "point": "GPT-6 Luna (agent loop) · Codex CLI",
          "metric": "agent-loop-tokens/Input tokens (cache reads included)",
          "label": "Tokens per attempt: single call vs agent loop (Input tokens (cache reads included))",
          "value": 15530,
          "unit": "tokens",
          "display": "15,530",
          "n": 14,
          "range": [
            15391,
            39009
          ],
          "spanKind": "minmax",
          "context": "Codex CLI · agent loop"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-tokens",
          "series": "Output tokens",
          "point": "GPT-6 Luna (single call) · Codex CLI",
          "metric": "agent-loop-tokens/Output tokens",
          "label": "Tokens per attempt: single call vs agent loop (Output tokens)",
          "value": 345,
          "unit": "tokens",
          "display": "345",
          "n": 16,
          "range": [
            36,
            634
          ],
          "spanKind": "minmax",
          "context": "Codex CLI · single call"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-tokens",
          "series": "Output tokens",
          "point": "GPT-6 Luna (agent loop) · Codex CLI",
          "metric": "agent-loop-tokens/Output tokens",
          "label": "Tokens per attempt: single call vs agent loop (Output tokens)",
          "value": 480,
          "unit": "tokens",
          "display": "480",
          "n": 14,
          "range": [
            143,
            858
          ],
          "spanKind": "minmax",
          "context": "Codex CLI · agent loop"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-tool-calls",
          "series": "Tool calls per attempt",
          "point": "GPT-6 Luna (agent loop) · Codex CLI",
          "metric": "agent-loop-tool-calls",
          "label": "Tool calls per agent-loop attempt",
          "value": 0,
          "unit": "count",
          "display": "0",
          "n": 14,
          "range": [
            0,
            1
          ],
          "spanKind": "minmax",
          "context": "Codex CLI · agent loop"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-cost-per-pass",
          "series": "Cost per strict pass",
          "point": "GPT-6 Luna (single call) · Codex CLI",
          "metric": "agent-loop-cost-per-pass",
          "label": "List-price cost per strict pass: single call vs agent loop (calculation)",
          "value": 0.00116,
          "unit": "usd",
          "display": "$0.0012",
          "n": 16,
          "calculation": true,
          "context": "Codex CLI · single call"
        },
        {
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-cost-per-pass",
          "series": "Cost per strict pass",
          "point": "GPT-6 Luna (agent loop) · Codex CLI",
          "metric": "agent-loop-cost-per-pass",
          "label": "List-price cost per strict pass: single call vs agent loop (calculation)",
          "value": 0.00099,
          "unit": "usd",
          "display": "$0.00099",
          "n": 14,
          "calculation": true,
          "context": "Codex CLI · agent loop"
        },
        {
          "studySlug": "llm-speed-anatomy",
          "chartId": "speed-anatomy-first-text",
          "series": "Time to first text",
          "point": "GPT-6 Luna (low) · Codex CLI",
          "metric": "speed-anatomy-first-text",
          "label": "Time to first text: a 250-line answer, six models",
          "value": 3.3,
          "unit": "seconds",
          "display": "3.30 s",
          "n": 4,
          "range": [
            3.19,
            3.47
          ],
          "spanKind": "minmax",
          "context": "Codex CLI · effort low"
        },
        {
          "studySlug": "llm-speed-anatomy",
          "chartId": "speed-anatomy-output-speed",
          "series": "Visible tokens per second",
          "point": "GPT-6 Luna (low) · Codex CLI",
          "metric": "speed-anatomy-output-speed",
          "label": "Output speed after the first text: visible tokens per second (calculation)",
          "value": 129.1,
          "unit": "tokens",
          "display": "129",
          "n": 4,
          "range": [
            55.5,
            259.1
          ],
          "spanKind": "minmax",
          "calculation": true,
          "context": "Codex CLI · effort low"
        },
        {
          "studySlug": "llm-speed-anatomy",
          "chartId": "speed-anatomy-chars-per-second",
          "series": "Characters per second",
          "point": "GPT-6 Luna (low) · Codex CLI",
          "metric": "speed-anatomy-chars-per-second",
          "label": "Output speed in characters per second after the first text (calculation)",
          "value": 524,
          "unit": "count",
          "display": "524",
          "n": 4,
          "range": [
            225,
            1052
          ],
          "spanKind": "minmax",
          "calculation": true,
          "context": "Codex CLI · effort low"
        }
      ]
    },
    {
      "slug": "gpt-6-luna-openai-api",
      "name": "GPT-6 Luna (OpenAI API)",
      "vendor": "OpenAI",
      "kind": "model",
      "description": "OpenAI’s GPT-6 Luna model called directly through the OpenAI API, without a CLI.",
      "aliases": [
        "GPT-6 Luna · OpenAI API"
      ],
      "facts": [
        {
          "studySlug": "cli-model-latency-tokens",
          "chartId": "cli-vs-api-exact-reply-latency",
          "series": "Total time",
          "point": "OpenAI API · GPT-6 Luna · none",
          "metric": "cli-vs-api-exact-reply-latency/Total time",
          "label": "CLI vs API: time for a one-line answer (Total time)",
          "value": 0.97,
          "unit": "seconds",
          "display": "0.97 s",
          "n": 5,
          "range": [
            0.65,
            1.5
          ],
          "spanKind": "minmax",
          "context": "OpenAI API · effort none · fixed exact reply, 5 runs"
        },
        {
          "studySlug": "cli-model-latency-tokens",
          "chartId": "cli-vs-api-exact-reply-latency",
          "series": "First useful output",
          "point": "OpenAI API · GPT-6 Luna · none",
          "metric": "cli-vs-api-exact-reply-latency/First useful output",
          "label": "CLI vs API: time for a one-line answer (First useful output)",
          "value": 0.82,
          "unit": "seconds",
          "display": "0.82 s",
          "n": 5,
          "range": [
            0.51,
            1.37
          ],
          "spanKind": "minmax",
          "context": "OpenAI API · effort none · fixed exact reply, 5 runs"
        },
        {
          "studySlug": "cli-model-latency-tokens",
          "chartId": "cli-vs-api-small-coding-latency",
          "series": "Total time",
          "point": "OpenAI API · GPT-6 Luna · none",
          "metric": "cli-vs-api-small-coding-latency/Total time",
          "label": "CLI vs API: time for a small coding task (Total time)",
          "value": 4.01,
          "unit": "seconds",
          "display": "4.01 s",
          "n": 3,
          "range": [
            3.83,
            4.35
          ],
          "spanKind": "minmax",
          "context": "OpenAI API · effort none · small coding task, 3 runs"
        },
        {
          "studySlug": "cli-model-latency-tokens",
          "chartId": "cli-vs-api-small-coding-latency",
          "series": "First useful output",
          "point": "OpenAI API · GPT-6 Luna · none",
          "metric": "cli-vs-api-small-coding-latency/First useful output",
          "label": "CLI vs API: time for a small coding task (First useful output)",
          "value": 0.67,
          "unit": "seconds",
          "display": "0.67 s",
          "n": 3,
          "range": [
            0.62,
            0.81
          ],
          "spanKind": "minmax",
          "context": "OpenAI API · effort none · small coding task, 3 runs"
        },
        {
          "studySlug": "cli-model-latency-tokens",
          "chartId": "cli-vs-api-prompt-overhead",
          "series": "Input tokens",
          "point": "OpenAI API · GPT-6 Luna · none",
          "metric": "cli-vs-api-prompt-overhead",
          "label": "Hidden prompt: input tokens for the same one-line request",
          "value": 17,
          "unit": "tokens",
          "display": "17",
          "n": 5,
          "context": "OpenAI API · effort none · short fixed tasks"
        }
      ]
    },
    {
      "slug": "gpt-5-2",
      "name": "GPT 5.2",
      "vendor": "OpenAI",
      "kind": "model",
      "description": "OpenAI model in the public SWE-bench Verified panel (mini-SWE-agent v2, high effort).",
      "aliases": [
        "GPT 5.2"
      ],
      "facts": [
        {
          "studySlug": "swe-bench-verified",
          "chartId": "swebench-same-instance-leaderboard",
          "series": "Resolved rate",
          "point": "GPT 5.2 (high)",
          "metric": "swebench-same-instance-leaderboard",
          "label": "Resolved rate on the same 33 SWE-bench Verified instances",
          "value": 0.8485,
          "unit": "rate",
          "display": "85% (28/33)",
          "n": 33,
          "ci": [
            0.6908,
            0.9335
          ],
          "spanKind": "ci95",
          "context": "effort high · public mini-SWE-agent v2 run, same instances"
        },
        {
          "studySlug": "swe-bench-verified",
          "chartId": "swebench-model-calls",
          "series": "Mean calls",
          "point": "GPT 5.2 (high)",
          "metric": "swebench-model-calls",
          "label": "Model calls per instance",
          "value": 35.6,
          "unit": "calls",
          "display": "35.6",
          "n": 33,
          "context": "effort high · public mini-SWE-agent v2 run, same instances"
        },
        {
          "studySlug": "cost-thought-experiments",
          "chartId": "cost-per-resolved-agent-vs-panel",
          "series": "Cost per resolved instance",
          "point": "GPT 5.2 (high)",
          "metric": "cost-per-resolved-agent-vs-panel",
          "label": "Recorded cost per resolved instance: Agent vs the public panel",
          "value": 0.628,
          "unit": "usd",
          "display": "$0.63",
          "n": 28,
          "context": "effort high · public mini-SWE-agent v2 run, same instances"
        }
      ]
    },
    {
      "slug": "gemini-3-flash",
      "name": "Gemini 3 Flash",
      "vendor": "Google",
      "kind": "model",
      "description": "Google model in the public SWE-bench Verified panel (mini-SWE-agent v2, high effort).",
      "aliases": [
        "Gemini 3 Flash"
      ],
      "facts": [
        {
          "studySlug": "swe-bench-verified",
          "chartId": "swebench-same-instance-leaderboard",
          "series": "Resolved rate",
          "point": "Gemini 3 Flash (high)",
          "metric": "swebench-same-instance-leaderboard",
          "label": "Resolved rate on the same 33 SWE-bench Verified instances",
          "value": 0.8182,
          "unit": "rate",
          "display": "82% (27/33)",
          "n": 33,
          "ci": [
            0.6561,
            0.9139
          ],
          "spanKind": "ci95",
          "context": "effort high · public mini-SWE-agent v2 run, same instances"
        },
        {
          "studySlug": "swe-bench-verified",
          "chartId": "swebench-model-calls",
          "series": "Mean calls",
          "point": "Gemini 3 Flash (high)",
          "metric": "swebench-model-calls",
          "label": "Model calls per instance",
          "value": 54.2,
          "unit": "calls",
          "display": "54.2",
          "n": 33,
          "context": "effort high · public mini-SWE-agent v2 run, same instances"
        },
        {
          "studySlug": "cost-thought-experiments",
          "chartId": "cost-per-resolved-agent-vs-panel",
          "series": "Cost per resolved instance",
          "point": "Gemini 3 Flash (high)",
          "metric": "cost-per-resolved-agent-vs-panel",
          "label": "Recorded cost per resolved instance: Agent vs the public panel",
          "value": 0.436,
          "unit": "usd",
          "display": "$0.44",
          "n": 27,
          "context": "effort high · public mini-SWE-agent v2 run, same instances"
        }
      ]
    },
    {
      "slug": "glm-5",
      "name": "GLM 5",
      "vendor": "Z.ai",
      "kind": "model",
      "description": "Z.ai model in the public SWE-bench Verified panel (mini-SWE-agent v2, high effort).",
      "aliases": [
        "GLM 5"
      ],
      "facts": [
        {
          "studySlug": "swe-bench-verified",
          "chartId": "swebench-same-instance-leaderboard",
          "series": "Resolved rate",
          "point": "GLM 5 (high)",
          "metric": "swebench-same-instance-leaderboard",
          "label": "Resolved rate on the same 33 SWE-bench Verified instances",
          "value": 0.7879,
          "unit": "rate",
          "display": "79% (26/33)",
          "n": 33,
          "ci": [
            0.6225,
            0.8932
          ],
          "spanKind": "ci95",
          "context": "effort high · public mini-SWE-agent v2 run, same instances"
        },
        {
          "studySlug": "swe-bench-verified",
          "chartId": "swebench-model-calls",
          "series": "Mean calls",
          "point": "GLM 5 (high)",
          "metric": "swebench-model-calls",
          "label": "Model calls per instance",
          "value": 77.5,
          "unit": "calls",
          "display": "77.5",
          "n": 33,
          "context": "effort high · public mini-SWE-agent v2 run, same instances"
        },
        {
          "studySlug": "cost-thought-experiments",
          "chartId": "cost-per-resolved-agent-vs-panel",
          "series": "Cost per resolved instance",
          "point": "GLM 5 (high)",
          "metric": "cost-per-resolved-agent-vs-panel",
          "label": "Recorded cost per resolved instance: Agent vs the public panel",
          "value": 0.667,
          "unit": "usd",
          "display": "$0.67",
          "n": 26,
          "context": "effort high · public mini-SWE-agent v2 run, same instances"
        }
      ]
    },
    {
      "slug": "claude-sonnet-4-5",
      "name": "Claude Sonnet 4.5",
      "vendor": "Anthropic",
      "kind": "model",
      "description": "Anthropic model in the public SWE-bench Verified panel (mini-SWE-agent v2, high effort).",
      "aliases": [
        "Claude Sonnet 4.5",
        "Claude 4.5 Sonnet"
      ],
      "facts": [
        {
          "studySlug": "swe-bench-verified",
          "chartId": "swebench-same-instance-leaderboard",
          "series": "Resolved rate",
          "point": "Claude 4.5 Sonnet (high)",
          "metric": "swebench-same-instance-leaderboard",
          "label": "Resolved rate on the same 33 SWE-bench Verified instances",
          "value": 0.7576,
          "unit": "rate",
          "display": "76% (25/33)",
          "n": 33,
          "ci": [
            0.5898,
            0.8717
          ],
          "spanKind": "ci95",
          "context": "effort high · public mini-SWE-agent v2 run, same instances"
        },
        {
          "studySlug": "swe-bench-verified",
          "chartId": "swebench-model-calls",
          "series": "Mean calls",
          "point": "Claude 4.5 Sonnet (high)",
          "metric": "swebench-model-calls",
          "label": "Model calls per instance",
          "value": 51,
          "unit": "calls",
          "display": "51",
          "n": 33,
          "context": "effort high · public mini-SWE-agent v2 run, same instances"
        },
        {
          "studySlug": "cost-thought-experiments",
          "chartId": "cost-per-resolved-agent-vs-panel",
          "series": "Cost per resolved instance",
          "point": "Claude 4.5 Sonnet (high)",
          "metric": "cost-per-resolved-agent-vs-panel",
          "label": "Recorded cost per resolved instance: Agent vs the public panel",
          "value": 0.913,
          "unit": "usd",
          "display": "$0.91",
          "n": 25,
          "context": "effort high · public mini-SWE-agent v2 run, same instances"
        }
      ]
    },
    {
      "slug": "claude-opus-4-5",
      "name": "Claude Opus 4.5",
      "vendor": "Anthropic",
      "kind": "model",
      "description": "Anthropic model in the public SWE-bench Verified panel (mini-SWE-agent v2, high effort).",
      "aliases": [
        "Claude Opus 4.5",
        "Claude 4.5 Opus"
      ],
      "facts": [
        {
          "studySlug": "swe-bench-verified",
          "chartId": "swebench-same-instance-leaderboard",
          "series": "Resolved rate",
          "point": "Claude 4.5 Opus (high)",
          "metric": "swebench-same-instance-leaderboard",
          "label": "Resolved rate on the same 33 SWE-bench Verified instances",
          "value": 0.7273,
          "unit": "rate",
          "display": "73% (24/33)",
          "n": 33,
          "ci": [
            0.5578,
            0.8493
          ],
          "spanKind": "ci95",
          "context": "effort high · public mini-SWE-agent v2 run, same instances"
        },
        {
          "studySlug": "swe-bench-verified",
          "chartId": "swebench-model-calls",
          "series": "Mean calls",
          "point": "Claude 4.5 Opus (high)",
          "metric": "swebench-model-calls",
          "label": "Model calls per instance",
          "value": 35.9,
          "unit": "calls",
          "display": "35.9",
          "n": 33,
          "context": "effort high · public mini-SWE-agent v2 run, same instances"
        },
        {
          "studySlug": "cost-thought-experiments",
          "chartId": "cost-per-resolved-agent-vs-panel",
          "series": "Cost per resolved instance",
          "point": "Claude 4.5 Opus (high)",
          "metric": "cost-per-resolved-agent-vs-panel",
          "label": "Recorded cost per resolved instance: Agent vs the public panel",
          "value": 1.184,
          "unit": "usd",
          "display": "$1.18",
          "n": 24,
          "context": "effort high · public mini-SWE-agent v2 run, same instances"
        }
      ]
    },
    {
      "slug": "claude-opus-4-6",
      "name": "Claude Opus 4.6",
      "vendor": "Anthropic",
      "kind": "model",
      "description": "Anthropic model in the public SWE-bench Verified panel (mini-SWE-agent v2).",
      "aliases": [
        "Claude Opus 4.6",
        "Claude 4.6 Opus"
      ],
      "facts": [
        {
          "studySlug": "swe-bench-verified",
          "chartId": "swebench-same-instance-leaderboard",
          "series": "Resolved rate",
          "point": "Claude 4.6 Opus",
          "metric": "swebench-same-instance-leaderboard",
          "label": "Resolved rate on the same 33 SWE-bench Verified instances",
          "value": 0.697,
          "unit": "rate",
          "display": "70% (23/33)",
          "n": 33,
          "ci": [
            0.5266,
            0.8262
          ],
          "spanKind": "ci95",
          "context": "public mini-SWE-agent v2 run, same instances"
        },
        {
          "studySlug": "swe-bench-verified",
          "chartId": "swebench-model-calls",
          "series": "Mean calls",
          "point": "Claude 4.6 Opus",
          "metric": "swebench-model-calls",
          "label": "Model calls per instance",
          "value": 28.9,
          "unit": "calls",
          "display": "28.9",
          "n": 33,
          "context": "public mini-SWE-agent v2 run, same instances"
        },
        {
          "studySlug": "cost-thought-experiments",
          "chartId": "cost-per-resolved-agent-vs-panel",
          "series": "Cost per resolved instance",
          "point": "Claude 4.6 Opus",
          "metric": "cost-per-resolved-agent-vs-panel",
          "label": "Recorded cost per resolved instance: Agent vs the public panel",
          "value": 0.875,
          "unit": "usd",
          "display": "$0.88",
          "n": 23,
          "context": "public mini-SWE-agent v2 run, same instances"
        }
      ]
    },
    {
      "slug": "deepseek-v3-2",
      "name": "DeepSeek V3.2",
      "vendor": "DeepSeek",
      "kind": "model",
      "description": "DeepSeek model in the public SWE-bench Verified panel (mini-SWE-agent v2, high effort).",
      "aliases": [
        "DeepSeek V3.2"
      ],
      "facts": [
        {
          "studySlug": "swe-bench-verified",
          "chartId": "swebench-same-instance-leaderboard",
          "series": "Resolved rate",
          "point": "DeepSeek V3.2 (high)",
          "metric": "swebench-same-instance-leaderboard",
          "label": "Resolved rate on the same 33 SWE-bench Verified instances",
          "value": 0.7273,
          "unit": "rate",
          "display": "73% (24/33)",
          "n": 33,
          "ci": [
            0.5578,
            0.8493
          ],
          "spanKind": "ci95",
          "context": "effort high · public mini-SWE-agent v2 run, same instances"
        },
        {
          "studySlug": "swe-bench-verified",
          "chartId": "swebench-model-calls",
          "series": "Mean calls",
          "point": "DeepSeek V3.2 (high)",
          "metric": "swebench-model-calls",
          "label": "Model calls per instance",
          "value": 88.2,
          "unit": "calls",
          "display": "88.2",
          "n": 33,
          "context": "effort high · public mini-SWE-agent v2 run, same instances"
        },
        {
          "studySlug": "cost-thought-experiments",
          "chartId": "cost-per-resolved-agent-vs-panel",
          "series": "Cost per resolved instance",
          "point": "DeepSeek V3.2 (high)",
          "metric": "cost-per-resolved-agent-vs-panel",
          "label": "Recorded cost per resolved instance: Agent vs the public panel",
          "value": 0.637,
          "unit": "usd",
          "display": "$0.64",
          "n": 24,
          "context": "effort high · public mini-SWE-agent v2 run, same instances"
        }
      ]
    },
    {
      "slug": "minimax-m2-5",
      "name": "MiniMax M2.5",
      "vendor": "MiniMax",
      "kind": "model",
      "description": "MiniMax model in the public SWE-bench Verified panel (mini-SWE-agent v2, high effort).",
      "aliases": [
        "MiniMax M2.5"
      ],
      "facts": [
        {
          "studySlug": "swe-bench-verified",
          "chartId": "swebench-same-instance-leaderboard",
          "series": "Resolved rate",
          "point": "MiniMax M2.5 (high)",
          "metric": "swebench-same-instance-leaderboard",
          "label": "Resolved rate on the same 33 SWE-bench Verified instances",
          "value": 0.697,
          "unit": "rate",
          "display": "70% (23/33)",
          "n": 33,
          "ci": [
            0.5266,
            0.8262
          ],
          "spanKind": "ci95",
          "context": "effort high · public mini-SWE-agent v2 run, same instances"
        },
        {
          "studySlug": "swe-bench-verified",
          "chartId": "swebench-model-calls",
          "series": "Mean calls",
          "point": "MiniMax M2.5 (high)",
          "metric": "swebench-model-calls",
          "label": "Model calls per instance",
          "value": 58.4,
          "unit": "calls",
          "display": "58.4",
          "n": 33,
          "context": "effort high · public mini-SWE-agent v2 run, same instances"
        },
        {
          "studySlug": "cost-thought-experiments",
          "chartId": "cost-per-resolved-agent-vs-panel",
          "series": "Cost per resolved instance",
          "point": "MiniMax M2.5 (high)",
          "metric": "cost-per-resolved-agent-vs-panel",
          "label": "Recorded cost per resolved instance: Agent vs the public panel",
          "value": 0.107,
          "unit": "usd",
          "display": "$0.11",
          "n": 23,
          "context": "effort high · public mini-SWE-agent v2 run, same instances"
        }
      ]
    },
    {
      "slug": "kimi-k2-5",
      "name": "Kimi K2.5",
      "vendor": "Moonshot AI",
      "kind": "model",
      "description": "Moonshot AI model in the public SWE-bench Verified panel (mini-SWE-agent v2, high effort).",
      "aliases": [
        "Kimi K2.5"
      ],
      "facts": [
        {
          "studySlug": "swe-bench-verified",
          "chartId": "swebench-same-instance-leaderboard",
          "series": "Resolved rate",
          "point": "Kimi K2.5 (high)",
          "metric": "swebench-same-instance-leaderboard",
          "label": "Resolved rate on the same 33 SWE-bench Verified instances",
          "value": 0.697,
          "unit": "rate",
          "display": "70% (23/33)",
          "n": 33,
          "ci": [
            0.5266,
            0.8262
          ],
          "spanKind": "ci95",
          "context": "effort high · public mini-SWE-agent v2 run, same instances"
        },
        {
          "studySlug": "swe-bench-verified",
          "chartId": "swebench-model-calls",
          "series": "Mean calls",
          "point": "Kimi K2.5 (high)",
          "metric": "swebench-model-calls",
          "label": "Model calls per instance",
          "value": 56.7,
          "unit": "calls",
          "display": "56.7",
          "n": 33,
          "context": "effort high · public mini-SWE-agent v2 run, same instances"
        },
        {
          "studySlug": "cost-thought-experiments",
          "chartId": "cost-per-resolved-agent-vs-panel",
          "series": "Cost per resolved instance",
          "point": "Kimi K2.5 (high)",
          "metric": "cost-per-resolved-agent-vs-panel",
          "label": "Recorded cost per resolved instance: Agent vs the public panel",
          "value": 0.256,
          "unit": "usd",
          "display": "$0.26",
          "n": 23,
          "context": "effort high · public mini-SWE-agent v2 run, same instances"
        }
      ]
    },
    {
      "slug": "gpt-5-mini",
      "name": "GPT 5 mini",
      "vendor": "OpenAI",
      "kind": "model",
      "description": "OpenAI model in the public SWE-bench Verified panel (mini-SWE-agent v2).",
      "aliases": [
        "GPT 5 mini"
      ],
      "facts": [
        {
          "studySlug": "swe-bench-verified",
          "chartId": "swebench-same-instance-leaderboard",
          "series": "Resolved rate",
          "point": "GPT 5 mini",
          "metric": "swebench-same-instance-leaderboard",
          "label": "Resolved rate on the same 33 SWE-bench Verified instances",
          "value": 0.6364,
          "unit": "rate",
          "display": "64% (21/33)",
          "n": 33,
          "ci": [
            0.4662,
            0.7781
          ],
          "spanKind": "ci95",
          "context": "public mini-SWE-agent v2 run, same instances"
        },
        {
          "studySlug": "swe-bench-verified",
          "chartId": "swebench-model-calls",
          "series": "Mean calls",
          "point": "GPT 5 mini",
          "metric": "swebench-model-calls",
          "label": "Model calls per instance",
          "value": 20.8,
          "unit": "calls",
          "display": "20.8",
          "n": 33,
          "context": "public mini-SWE-agent v2 run, same instances"
        },
        {
          "studySlug": "cost-thought-experiments",
          "chartId": "cost-per-resolved-agent-vs-panel",
          "series": "Cost per resolved instance",
          "point": "GPT 5 mini",
          "metric": "cost-per-resolved-agent-vs-panel",
          "label": "Recorded cost per resolved instance: Agent vs the public panel",
          "value": 0.08,
          "unit": "usd",
          "display": "$0.080",
          "n": 21,
          "context": "public mini-SWE-agent v2 run, same instances"
        }
      ]
    },
    {
      "slug": "claude-opus-5",
      "name": "Claude Opus 5",
      "vendor": "Anthropic",
      "kind": "model",
      "description": "Anthropic model, measured as a blind code-review critic and in a list-price calculation.",
      "aliases": [
        "Claude Opus 5"
      ],
      "facts": [
        {
          "studySlug": "blind-review-head-to-head",
          "chartId": "blind-review-critic-agreement",
          "series": "Critic model",
          "point": "Claude Opus 5 (Anthropic)",
          "metric": "blind-review-critic-agreement",
          "label": "Does the judge’s model family matter?",
          "value": 0.7,
          "unit": "rate",
          "display": "70% (28/40)",
          "n": 40,
          "ci": [
            0.5457,
            0.8193
          ],
          "spanKind": "ci95",
          "context": "blind review panel"
        },
        {
          "studySlug": "cost-thought-experiments",
          "chartId": "repriced-cost-per-resolved",
          "series": "Repriced cost per resolved instance",
          "point": "Claude Opus 5",
          "metric": "repriced-cost-per-resolved",
          "label": "Thought experiment: the same tokens at other list prices",
          "value": 8.723,
          "unit": "usd",
          "display": "$8.72",
          "calculation": true,
          "context": "calculation: Agent’s recorded tokens at this model’s list price"
        }
      ]
    },
    {
      "slug": "claude-fable-5",
      "name": "Claude Fable 5",
      "vendor": "Anthropic",
      "kind": "model",
      "description": "Anthropic model, measured as a blind code-review critic.",
      "aliases": [
        "Claude Fable 5"
      ],
      "facts": [
        {
          "studySlug": "blind-review-head-to-head",
          "chartId": "blind-review-critic-agreement",
          "series": "Critic model",
          "point": "Claude Fable 5 (Anthropic)",
          "metric": "blind-review-critic-agreement",
          "label": "Does the judge’s model family matter?",
          "value": 0.65,
          "unit": "rate",
          "display": "65% (26/40)",
          "n": 40,
          "ci": [
            0.4951,
            0.7787
          ],
          "spanKind": "ci95",
          "context": "blind review panel"
        }
      ]
    },
    {
      "slug": "gpt-5-5",
      "name": "GPT 5.5",
      "vendor": "OpenAI",
      "kind": "model",
      "description": "OpenAI model, measured as a blind code-review critic on a small number of pairs.",
      "aliases": [
        "GPT 5.5"
      ],
      "facts": [
        {
          "studySlug": "blind-review-head-to-head",
          "chartId": "blind-review-critic-agreement",
          "series": "Critic model",
          "point": "GPT 5.5 (OpenAI)",
          "metric": "blind-review-critic-agreement",
          "label": "Does the judge’s model family matter?",
          "value": 1,
          "unit": "rate",
          "display": "100% (2/2)",
          "n": 2,
          "ci": [
            0.3424,
            1
          ],
          "spanKind": "ci95",
          "context": "blind review panel"
        }
      ]
    },
    {
      "slug": "deterministic-routing-policy",
      "name": "Deterministic routing policy",
      "vendor": "Agent",
      "kind": "router",
      "description": "Agent’s rule-based routing: an in-process policy picks the model and effort for each call from the task stage and signals. No model call, so no token cost.",
      "aliases": [
        "Deterministic routing policy"
      ],
      "facts": [
        {
          "studySlug": "routing-overhead",
          "chartId": "router-overhead-decision-latency",
          "series": "Decision time",
          "point": "Deterministic routing policy (Agent, in process)",
          "metric": "router-overhead-decision-latency",
          "label": "Time to make one routing decision",
          "value": 0.00142,
          "unit": "ms",
          "display": "1.42 µs",
          "n": 20000,
          "range": [
            0.00142,
            0.00233
          ],
          "spanKind": "p50-p95",
          "context": "Agent · in process · routing overhead per decision"
        },
        {
          "studySlug": "routing-overhead",
          "chartId": "router-overhead-completed",
          "series": "Completed",
          "point": "Deterministic routing policy (Agent, in process)",
          "metric": "router-overhead-completed",
          "label": "Routing calls that returned a decision",
          "value": 1,
          "unit": "rate",
          "display": "100% (20000/20000)",
          "n": 20000,
          "ci": [
            0.9998,
            1
          ],
          "spanKind": "ci95",
          "context": "Agent · in process · routing overhead per decision"
        },
        {
          "studySlug": "routing-overhead",
          "chartId": "router-overhead-cost-reported",
          "series": "Cost per 1,000 decisions",
          "point": "Deterministic routing policy (Agent, in process)",
          "metric": "router-overhead-cost-reported",
          "label": "Cost per 1,000 routing decisions: no model call vs provider-reported",
          "value": 0,
          "unit": "usd",
          "display": "$0.00",
          "n": 20000,
          "context": "Agent · in process · routing overhead per decision"
        },
        {
          "studySlug": "routing-overhead",
          "chartId": "router-overhead-cost-per-1000-tasks",
          "series": "Every model call routed (49.5 per task)",
          "point": "Deterministic routing policy (Agent, in process)",
          "metric": "router-overhead-cost-per-1000-tasks/Every model call routed (49.5 per task)",
          "label": "Added routing cost per 1,000 tasks (calculation) (Every model call routed (49.5 per task))",
          "value": 0,
          "unit": "usd",
          "display": "$0.00",
          "calculation": true,
          "context": "Agent · in process · calculation per 1,000 tasks from recorded decision counts"
        },
        {
          "studySlug": "routing-overhead",
          "chartId": "router-overhead-cost-per-1000-tasks",
          "series": "Only System One decisions (7 per task)",
          "point": "Deterministic routing policy (Agent, in process)",
          "metric": "router-overhead-cost-per-1000-tasks/Only System One decisions (7 per task)",
          "label": "Added routing cost per 1,000 tasks (calculation) (Only System One decisions (7 per task))",
          "value": 0,
          "unit": "usd",
          "display": "$0.00",
          "calculation": true,
          "context": "Agent · in process · calculation per 1,000 tasks from recorded decision counts"
        },
        {
          "studySlug": "routing-overhead",
          "chartId": "router-overhead-delay-per-task",
          "series": "Every model call routed (49.5 per task)",
          "point": "Deterministic routing policy (Agent, in process)",
          "metric": "router-overhead-delay-per-task/Every model call routed (49.5 per task)",
          "label": "Added routing delay per task (calculation) (Every model call routed (49.5 per task))",
          "value": 0.0000703,
          "unit": "seconds",
          "display": "70.3 µs",
          "calculation": true,
          "context": "Agent · in process · calculation per task from recorded decision counts, decisions in line"
        },
        {
          "studySlug": "routing-overhead",
          "chartId": "router-overhead-delay-per-task",
          "series": "Only System One decisions (7 per task)",
          "point": "Deterministic routing policy (Agent, in process)",
          "metric": "router-overhead-delay-per-task/Only System One decisions (7 per task)",
          "label": "Added routing delay per task (calculation) (Only System One decisions (7 per task))",
          "value": 0.0000099,
          "unit": "seconds",
          "display": "9.9 µs",
          "calculation": true,
          "context": "Agent · in process · calculation per task from recorded decision counts, decisions in line"
        }
      ]
    },
    {
      "slug": "openrouter",
      "name": "OpenRouter",
      "vendor": "OpenRouter",
      "kind": "provider",
      "description": "A gateway that routes one API to many inference providers. Its per-token price is compared with first-party list prices; its fee is charged when credits are bought.",
      "aliases": [
        "OpenRouter"
      ],
      "facts": [
        {
          "studySlug": "inference-provider-index",
          "chartId": "gateway-vs-direct-claude-haiku-4-5",
          "series": "Input",
          "point": "OpenRouter",
          "metric": "gateway-vs-direct-claude-haiku-4-5/Input",
          "label": "Claude Haiku 4.5: OpenRouter vs Anthropic list price (Input)",
          "value": 1,
          "unit": "usd",
          "display": "$1.00",
          "context": "Claude Haiku 4.5 · list price, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "gateway-vs-direct-claude-haiku-4-5",
          "series": "Output",
          "point": "OpenRouter",
          "metric": "gateway-vs-direct-claude-haiku-4-5/Output",
          "label": "Claude Haiku 4.5: OpenRouter vs Anthropic list price (Output)",
          "value": 5,
          "unit": "usd",
          "display": "$5.00",
          "context": "Claude Haiku 4.5 · list price, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "gateway-vs-direct-claude-sonnet-5",
          "series": "Input",
          "point": "OpenRouter",
          "metric": "gateway-vs-direct-claude-sonnet-5/Input",
          "label": "Claude Sonnet 5: OpenRouter vs Anthropic list price (Input)",
          "value": 2,
          "unit": "usd",
          "display": "$2.00",
          "context": "Claude Sonnet 5 · list price, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "gateway-vs-direct-claude-sonnet-5",
          "series": "Output",
          "point": "OpenRouter",
          "metric": "gateway-vs-direct-claude-sonnet-5/Output",
          "label": "Claude Sonnet 5: OpenRouter vs Anthropic list price (Output)",
          "value": 10,
          "unit": "usd",
          "display": "$10.00",
          "context": "Claude Sonnet 5 · list price, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "gateway-vs-direct-claude-sonnet-5-5",
          "series": "Input",
          "point": "OpenRouter",
          "metric": "gateway-vs-direct-claude-sonnet-5-5/Input",
          "label": "Claude Sonnet 5.5: OpenRouter vs Anthropic list price (Input)",
          "value": 2,
          "unit": "usd",
          "display": "$2.00",
          "context": "Claude Sonnet 5.5 · list price, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "gateway-vs-direct-claude-sonnet-5-5",
          "series": "Output",
          "point": "OpenRouter",
          "metric": "gateway-vs-direct-claude-sonnet-5-5/Output",
          "label": "Claude Sonnet 5.5: OpenRouter vs Anthropic list price (Output)",
          "value": 10,
          "unit": "usd",
          "display": "$10.00",
          "context": "Claude Sonnet 5.5 · list price, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "gateway-vs-direct-claude-opus-4-8",
          "series": "Input",
          "point": "OpenRouter",
          "metric": "gateway-vs-direct-claude-opus-4-8/Input",
          "label": "Claude Opus 4.8: OpenRouter vs Anthropic list price (Input)",
          "value": 5,
          "unit": "usd",
          "display": "$5.00",
          "context": "Claude Opus 4.8 · list price, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "gateway-vs-direct-claude-opus-4-8",
          "series": "Output",
          "point": "OpenRouter",
          "metric": "gateway-vs-direct-claude-opus-4-8/Output",
          "label": "Claude Opus 4.8: OpenRouter vs Anthropic list price (Output)",
          "value": 25,
          "unit": "usd",
          "display": "$25.00",
          "context": "Claude Opus 4.8 · list price, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "gateway-vs-direct-claude-opus-5",
          "series": "Input",
          "point": "OpenRouter",
          "metric": "gateway-vs-direct-claude-opus-5/Input",
          "label": "Claude Opus 5: OpenRouter vs Anthropic list price (Input)",
          "value": 5,
          "unit": "usd",
          "display": "$5.00",
          "context": "Claude Opus 5 · list price, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "gateway-vs-direct-claude-opus-5",
          "series": "Output",
          "point": "OpenRouter",
          "metric": "gateway-vs-direct-claude-opus-5/Output",
          "label": "Claude Opus 5: OpenRouter vs Anthropic list price (Output)",
          "value": 25,
          "unit": "usd",
          "display": "$25.00",
          "context": "Claude Opus 5 · list price, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "gateway-vs-direct-claude-opus-5-5",
          "series": "Input",
          "point": "OpenRouter",
          "metric": "gateway-vs-direct-claude-opus-5-5/Input",
          "label": "Claude Opus 5.5: OpenRouter vs Anthropic list price (Input)",
          "value": 4,
          "unit": "usd",
          "display": "$4.00",
          "context": "Claude Opus 5.5 · list price, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "gateway-vs-direct-claude-opus-5-5",
          "series": "Output",
          "point": "OpenRouter",
          "metric": "gateway-vs-direct-claude-opus-5-5/Output",
          "label": "Claude Opus 5.5: OpenRouter vs Anthropic list price (Output)",
          "value": 20,
          "unit": "usd",
          "display": "$20.00",
          "context": "Claude Opus 5.5 · list price, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "gateway-vs-direct-claude-fable-5-1",
          "series": "Input",
          "point": "OpenRouter",
          "metric": "gateway-vs-direct-claude-fable-5-1/Input",
          "label": "Claude Fable 5.1: OpenRouter vs Anthropic list price (Input)",
          "value": 10,
          "unit": "usd",
          "display": "$10.00",
          "context": "Claude Fable 5.1 · list price, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "gateway-vs-direct-claude-fable-5-1",
          "series": "Output",
          "point": "OpenRouter",
          "metric": "gateway-vs-direct-claude-fable-5-1/Output",
          "label": "Claude Fable 5.1: OpenRouter vs Anthropic list price (Output)",
          "value": 50,
          "unit": "usd",
          "display": "$50.00",
          "context": "Claude Fable 5.1 · list price, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "gateway-vs-direct-gpt-6-luna",
          "series": "Input",
          "point": "OpenRouter",
          "metric": "gateway-vs-direct-gpt-6-luna/Input",
          "label": "GPT-6 Luna: OpenRouter vs OpenAI list price (Input)",
          "value": 0.1,
          "unit": "usd",
          "display": "$0.10",
          "context": "GPT-6 Luna · list price, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "gateway-vs-direct-gpt-6-luna",
          "series": "Output",
          "point": "OpenRouter",
          "metric": "gateway-vs-direct-gpt-6-luna/Output",
          "label": "GPT-6 Luna: OpenRouter vs OpenAI list price (Output)",
          "value": 0.5,
          "unit": "usd",
          "display": "$0.50",
          "context": "GPT-6 Luna · list price, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "gateway-vs-direct-gemini-3-8-flash",
          "series": "Input",
          "point": "OpenRouter",
          "metric": "gateway-vs-direct-gemini-3-8-flash/Input",
          "label": "Gemini 3.8 Flash: OpenRouter vs Google list price (Input)",
          "value": 0.75,
          "unit": "usd",
          "display": "$0.75",
          "context": "Gemini 3.8 Flash · list price, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "gateway-vs-direct-gemini-3-8-flash",
          "series": "Output",
          "point": "OpenRouter",
          "metric": "gateway-vs-direct-gemini-3-8-flash/Output",
          "label": "Gemini 3.8 Flash: OpenRouter vs Google list price (Output)",
          "value": 3.75,
          "unit": "usd",
          "display": "$3.75",
          "context": "Gemini 3.8 Flash · list price, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "gateway-vs-direct-gemini-3-5-flash",
          "series": "Input",
          "point": "OpenRouter",
          "metric": "gateway-vs-direct-gemini-3-5-flash/Input",
          "label": "Gemini 3.5 Flash: OpenRouter vs Google list price (Input)",
          "value": 1.5,
          "unit": "usd",
          "display": "$1.50",
          "context": "Gemini 3.5 Flash · list price, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "gateway-vs-direct-gemini-3-5-flash",
          "series": "Output",
          "point": "OpenRouter",
          "metric": "gateway-vs-direct-gemini-3-5-flash/Output",
          "label": "Gemini 3.5 Flash: OpenRouter vs Google list price (Output)",
          "value": 9,
          "unit": "usd",
          "display": "$9.00",
          "context": "Gemini 3.5 Flash · list price, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "anthropic",
      "name": "Anthropic",
      "vendor": "Anthropic",
      "kind": "provider",
      "description": "Anthropic’s own API: the first-party list price of the Claude models, and Anthropic’s endpoint as listed on OpenRouter.",
      "aliases": [
        "Anthropic"
      ],
      "facts": [
        {
          "studySlug": "inference-provider-index",
          "chartId": "gateway-vs-direct-claude-haiku-4-5",
          "series": "Input",
          "point": "Anthropic (first-party list price)",
          "metric": "gateway-vs-direct-claude-haiku-4-5/Input",
          "label": "Claude Haiku 4.5: OpenRouter vs Anthropic list price (Input)",
          "value": 1,
          "unit": "usd",
          "display": "$1.00",
          "context": "first-party list price · Claude Haiku 4.5 · list price, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "gateway-vs-direct-claude-haiku-4-5",
          "series": "Output",
          "point": "Anthropic (first-party list price)",
          "metric": "gateway-vs-direct-claude-haiku-4-5/Output",
          "label": "Claude Haiku 4.5: OpenRouter vs Anthropic list price (Output)",
          "value": 5,
          "unit": "usd",
          "display": "$5.00",
          "context": "first-party list price · Claude Haiku 4.5 · list price, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "gateway-vs-direct-claude-sonnet-5",
          "series": "Input",
          "point": "Anthropic (first-party list price)",
          "metric": "gateway-vs-direct-claude-sonnet-5/Input",
          "label": "Claude Sonnet 5: OpenRouter vs Anthropic list price (Input)",
          "value": 2,
          "unit": "usd",
          "display": "$2.00",
          "context": "first-party list price · Claude Sonnet 5 · list price, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "gateway-vs-direct-claude-sonnet-5",
          "series": "Output",
          "point": "Anthropic (first-party list price)",
          "metric": "gateway-vs-direct-claude-sonnet-5/Output",
          "label": "Claude Sonnet 5: OpenRouter vs Anthropic list price (Output)",
          "value": 10,
          "unit": "usd",
          "display": "$10.00",
          "context": "first-party list price · Claude Sonnet 5 · list price, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "gateway-vs-direct-claude-sonnet-5-5",
          "series": "Input",
          "point": "Anthropic (first-party list price)",
          "metric": "gateway-vs-direct-claude-sonnet-5-5/Input",
          "label": "Claude Sonnet 5.5: OpenRouter vs Anthropic list price (Input)",
          "value": 2,
          "unit": "usd",
          "display": "$2.00",
          "context": "first-party list price · Claude Sonnet 5.5 · list price, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "gateway-vs-direct-claude-sonnet-5-5",
          "series": "Output",
          "point": "Anthropic (first-party list price)",
          "metric": "gateway-vs-direct-claude-sonnet-5-5/Output",
          "label": "Claude Sonnet 5.5: OpenRouter vs Anthropic list price (Output)",
          "value": 10,
          "unit": "usd",
          "display": "$10.00",
          "context": "first-party list price · Claude Sonnet 5.5 · list price, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "gateway-vs-direct-claude-opus-4-8",
          "series": "Input",
          "point": "Anthropic (first-party list price)",
          "metric": "gateway-vs-direct-claude-opus-4-8/Input",
          "label": "Claude Opus 4.8: OpenRouter vs Anthropic list price (Input)",
          "value": 5,
          "unit": "usd",
          "display": "$5.00",
          "context": "first-party list price · Claude Opus 4.8 · list price, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "gateway-vs-direct-claude-opus-4-8",
          "series": "Output",
          "point": "Anthropic (first-party list price)",
          "metric": "gateway-vs-direct-claude-opus-4-8/Output",
          "label": "Claude Opus 4.8: OpenRouter vs Anthropic list price (Output)",
          "value": 25,
          "unit": "usd",
          "display": "$25.00",
          "context": "first-party list price · Claude Opus 4.8 · list price, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "gateway-vs-direct-claude-opus-5",
          "series": "Input",
          "point": "Anthropic (first-party list price)",
          "metric": "gateway-vs-direct-claude-opus-5/Input",
          "label": "Claude Opus 5: OpenRouter vs Anthropic list price (Input)",
          "value": 5,
          "unit": "usd",
          "display": "$5.00",
          "context": "first-party list price · Claude Opus 5 · list price, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "gateway-vs-direct-claude-opus-5",
          "series": "Output",
          "point": "Anthropic (first-party list price)",
          "metric": "gateway-vs-direct-claude-opus-5/Output",
          "label": "Claude Opus 5: OpenRouter vs Anthropic list price (Output)",
          "value": 25,
          "unit": "usd",
          "display": "$25.00",
          "context": "first-party list price · Claude Opus 5 · list price, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "gateway-vs-direct-claude-opus-5-5",
          "series": "Input",
          "point": "Anthropic (first-party list price)",
          "metric": "gateway-vs-direct-claude-opus-5-5/Input",
          "label": "Claude Opus 5.5: OpenRouter vs Anthropic list price (Input)",
          "value": 4,
          "unit": "usd",
          "display": "$4.00",
          "context": "first-party list price · Claude Opus 5.5 · list price, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "gateway-vs-direct-claude-opus-5-5",
          "series": "Output",
          "point": "Anthropic (first-party list price)",
          "metric": "gateway-vs-direct-claude-opus-5-5/Output",
          "label": "Claude Opus 5.5: OpenRouter vs Anthropic list price (Output)",
          "value": 20,
          "unit": "usd",
          "display": "$20.00",
          "context": "first-party list price · Claude Opus 5.5 · list price, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "gateway-vs-direct-claude-fable-5-1",
          "series": "Input",
          "point": "Anthropic (first-party list price)",
          "metric": "gateway-vs-direct-claude-fable-5-1/Input",
          "label": "Claude Fable 5.1: OpenRouter vs Anthropic list price (Input)",
          "value": 10,
          "unit": "usd",
          "display": "$10.00",
          "context": "first-party list price · Claude Fable 5.1 · list price, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "gateway-vs-direct-claude-fable-5-1",
          "series": "Output",
          "point": "Anthropic (first-party list price)",
          "metric": "gateway-vs-direct-claude-fable-5-1/Output",
          "label": "Claude Fable 5.1: OpenRouter vs Anthropic list price (Output)",
          "value": 50,
          "unit": "usd",
          "display": "$50.00",
          "context": "first-party list price · Claude Fable 5.1 · list price, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-haiku-4-5",
          "series": "Input",
          "point": "Anthropic",
          "metric": "provider-prices-claude-haiku-4-5/Input",
          "label": "Claude Haiku 4.5: price per million tokens by provider (Input)",
          "value": 1,
          "unit": "usd",
          "display": "$1.00",
          "context": "Claude Haiku 4.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-haiku-4-5",
          "series": "Output",
          "point": "Anthropic",
          "metric": "provider-prices-claude-haiku-4-5/Output",
          "label": "Claude Haiku 4.5: price per million tokens by provider (Output)",
          "value": 5,
          "unit": "usd",
          "display": "$5.00",
          "context": "Claude Haiku 4.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-haiku-4-5",
          "series": "Cache read",
          "point": "Anthropic",
          "metric": "provider-prices-claude-haiku-4-5/Cache read",
          "label": "Claude Haiku 4.5: price per million tokens by provider (Cache read)",
          "value": 0.1,
          "unit": "usd",
          "display": "$0.10",
          "context": "Claude Haiku 4.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-sonnet-5",
          "series": "Input",
          "point": "Anthropic",
          "metric": "provider-prices-claude-sonnet-5/Input",
          "label": "Claude Sonnet 5: price per million tokens by provider (Input)",
          "value": 2,
          "unit": "usd",
          "display": "$2.00",
          "context": "Claude Sonnet 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-sonnet-5",
          "series": "Output",
          "point": "Anthropic",
          "metric": "provider-prices-claude-sonnet-5/Output",
          "label": "Claude Sonnet 5: price per million tokens by provider (Output)",
          "value": 10,
          "unit": "usd",
          "display": "$10.00",
          "context": "Claude Sonnet 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-sonnet-5",
          "series": "Cache read",
          "point": "Anthropic",
          "metric": "provider-prices-claude-sonnet-5/Cache read",
          "label": "Claude Sonnet 5: price per million tokens by provider (Cache read)",
          "value": 0.2,
          "unit": "usd",
          "display": "$0.20",
          "context": "Claude Sonnet 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-sonnet-5-5",
          "series": "Input",
          "point": "Anthropic",
          "metric": "provider-prices-claude-sonnet-5-5/Input",
          "label": "Claude Sonnet 5.5: price per million tokens by provider (Input)",
          "value": 2,
          "unit": "usd",
          "display": "$2.00",
          "context": "Claude Sonnet 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-sonnet-5-5",
          "series": "Output",
          "point": "Anthropic",
          "metric": "provider-prices-claude-sonnet-5-5/Output",
          "label": "Claude Sonnet 5.5: price per million tokens by provider (Output)",
          "value": 10,
          "unit": "usd",
          "display": "$10.00",
          "context": "Claude Sonnet 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-sonnet-5-5",
          "series": "Cache read",
          "point": "Anthropic",
          "metric": "provider-prices-claude-sonnet-5-5/Cache read",
          "label": "Claude Sonnet 5.5: price per million tokens by provider (Cache read)",
          "value": 0.2,
          "unit": "usd",
          "display": "$0.20",
          "context": "Claude Sonnet 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-4-8",
          "series": "Input",
          "point": "Anthropic",
          "metric": "provider-prices-claude-opus-4-8/Input",
          "label": "Claude Opus 4.8: price per million tokens by provider (Input)",
          "value": 5,
          "unit": "usd",
          "display": "$5.00",
          "context": "Claude Opus 4.8 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-4-8",
          "series": "Output",
          "point": "Anthropic",
          "metric": "provider-prices-claude-opus-4-8/Output",
          "label": "Claude Opus 4.8: price per million tokens by provider (Output)",
          "value": 25,
          "unit": "usd",
          "display": "$25.00",
          "context": "Claude Opus 4.8 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-4-8",
          "series": "Cache read",
          "point": "Anthropic",
          "metric": "provider-prices-claude-opus-4-8/Cache read",
          "label": "Claude Opus 4.8: price per million tokens by provider (Cache read)",
          "value": 0.5,
          "unit": "usd",
          "display": "$0.50",
          "context": "Claude Opus 4.8 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-5",
          "series": "Input",
          "point": "Anthropic",
          "metric": "provider-prices-claude-opus-5/Input",
          "label": "Claude Opus 5: price per million tokens by provider (Input)",
          "value": 5,
          "unit": "usd",
          "display": "$5.00",
          "context": "Claude Opus 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-5",
          "series": "Output",
          "point": "Anthropic",
          "metric": "provider-prices-claude-opus-5/Output",
          "label": "Claude Opus 5: price per million tokens by provider (Output)",
          "value": 25,
          "unit": "usd",
          "display": "$25.00",
          "context": "Claude Opus 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-5",
          "series": "Cache read",
          "point": "Anthropic",
          "metric": "provider-prices-claude-opus-5/Cache read",
          "label": "Claude Opus 5: price per million tokens by provider (Cache read)",
          "value": 0.5,
          "unit": "usd",
          "display": "$0.50",
          "context": "Claude Opus 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-5-5",
          "series": "Input",
          "point": "Anthropic",
          "metric": "provider-prices-claude-opus-5-5/Input",
          "label": "Claude Opus 5.5: price per million tokens by provider (Input)",
          "value": 4,
          "unit": "usd",
          "display": "$4.00",
          "context": "Claude Opus 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-5-5",
          "series": "Output",
          "point": "Anthropic",
          "metric": "provider-prices-claude-opus-5-5/Output",
          "label": "Claude Opus 5.5: price per million tokens by provider (Output)",
          "value": 20,
          "unit": "usd",
          "display": "$20.00",
          "context": "Claude Opus 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-5-5",
          "series": "Cache read",
          "point": "Anthropic",
          "metric": "provider-prices-claude-opus-5-5/Cache read",
          "label": "Claude Opus 5.5: price per million tokens by provider (Cache read)",
          "value": 0.2,
          "unit": "usd",
          "display": "$0.20",
          "context": "Claude Opus 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-fable-5-1",
          "series": "Input",
          "point": "Anthropic",
          "metric": "provider-prices-claude-fable-5-1/Input",
          "label": "Claude Fable 5.1: price per million tokens by provider (Input)",
          "value": 10,
          "unit": "usd",
          "display": "$10.00",
          "context": "Claude Fable 5.1 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-fable-5-1",
          "series": "Output",
          "point": "Anthropic",
          "metric": "provider-prices-claude-fable-5-1/Output",
          "label": "Claude Fable 5.1: price per million tokens by provider (Output)",
          "value": 50,
          "unit": "usd",
          "display": "$50.00",
          "context": "Claude Fable 5.1 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-fable-5-1",
          "series": "Cache read",
          "point": "Anthropic",
          "metric": "provider-prices-claude-fable-5-1/Cache read",
          "label": "Claude Fable 5.1: price per million tokens by provider (Cache read)",
          "value": 0.25,
          "unit": "usd",
          "display": "$0.25",
          "context": "Claude Fable 5.1 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "openai",
      "name": "OpenAI",
      "vendor": "OpenAI",
      "kind": "provider",
      "description": "OpenAI’s own API: the first-party list price of GPT models, and OpenAI’s endpoints as listed on OpenRouter.",
      "aliases": [
        "OpenAI"
      ],
      "facts": [
        {
          "studySlug": "inference-provider-index",
          "chartId": "gateway-vs-direct-gpt-6-luna",
          "series": "Input",
          "point": "OpenAI (first-party list price)",
          "metric": "gateway-vs-direct-gpt-6-luna/Input",
          "label": "GPT-6 Luna: OpenRouter vs OpenAI list price (Input)",
          "value": 0.1,
          "unit": "usd",
          "display": "$0.10",
          "context": "first-party list price · GPT-6 Luna · list price, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "gateway-vs-direct-gpt-6-luna",
          "series": "Output",
          "point": "OpenAI (first-party list price)",
          "metric": "gateway-vs-direct-gpt-6-luna/Output",
          "label": "GPT-6 Luna: OpenRouter vs OpenAI list price (Output)",
          "value": 0.5,
          "unit": "usd",
          "display": "$0.50",
          "context": "first-party list price · GPT-6 Luna · list price, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-6-sol",
          "series": "Input",
          "point": "OpenAI",
          "metric": "provider-prices-gpt-6-sol/Input",
          "label": "GPT-6 Sol: price per million tokens by provider (Input)",
          "value": 2,
          "unit": "usd",
          "display": "$2.00",
          "context": "GPT-6 Sol · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-6-sol",
          "series": "Output",
          "point": "OpenAI",
          "metric": "provider-prices-gpt-6-sol/Output",
          "label": "GPT-6 Sol: price per million tokens by provider (Output)",
          "value": 10,
          "unit": "usd",
          "display": "$10.00",
          "context": "GPT-6 Sol · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-6-sol",
          "series": "Cache read",
          "point": "OpenAI",
          "metric": "provider-prices-gpt-6-sol/Cache read",
          "label": "GPT-6 Sol: price per million tokens by provider (Cache read)",
          "value": 0.2,
          "unit": "usd",
          "display": "$0.20",
          "context": "GPT-6 Sol · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-6-luna",
          "series": "Input",
          "point": "OpenAI",
          "metric": "provider-prices-gpt-6-luna/Input",
          "label": "GPT-6 Luna: price per million tokens by provider (Input)",
          "value": 0.1,
          "unit": "usd",
          "display": "$0.10",
          "context": "GPT-6 Luna · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-6-luna",
          "series": "Output",
          "point": "OpenAI",
          "metric": "provider-prices-gpt-6-luna/Output",
          "label": "GPT-6 Luna: price per million tokens by provider (Output)",
          "value": 0.5,
          "unit": "usd",
          "display": "$0.50",
          "context": "GPT-6 Luna · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-6-luna",
          "series": "Cache read",
          "point": "OpenAI",
          "metric": "provider-prices-gpt-6-luna/Cache read",
          "label": "GPT-6 Luna: price per million tokens by provider (Cache read)",
          "value": 0.01,
          "unit": "usd",
          "display": "$0.010",
          "context": "GPT-6 Luna · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-6-astra",
          "series": "Input",
          "point": "OpenAI",
          "metric": "provider-prices-gpt-6-astra/Input",
          "label": "GPT-6 Astra: price per million tokens by provider (Input)",
          "value": 10,
          "unit": "usd",
          "display": "$10.00",
          "context": "GPT-6 Astra · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-6-astra",
          "series": "Output",
          "point": "OpenAI",
          "metric": "provider-prices-gpt-6-astra/Output",
          "label": "GPT-6 Astra: price per million tokens by provider (Output)",
          "value": 50,
          "unit": "usd",
          "display": "$50.00",
          "context": "GPT-6 Astra · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-6-astra",
          "series": "Cache read",
          "point": "OpenAI",
          "metric": "provider-prices-gpt-6-astra/Cache read",
          "label": "GPT-6 Astra: price per million tokens by provider (Cache read)",
          "value": 1,
          "unit": "usd",
          "display": "$1.00",
          "context": "GPT-6 Astra · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-5-5",
          "series": "Input",
          "point": "OpenAI",
          "metric": "provider-prices-gpt-5-5/Input",
          "label": "GPT-5.5: price per million tokens by provider (Input)",
          "value": 5,
          "unit": "usd",
          "display": "$5.00",
          "context": "GPT-5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-5-5",
          "series": "Output",
          "point": "OpenAI",
          "metric": "provider-prices-gpt-5-5/Output",
          "label": "GPT-5.5: price per million tokens by provider (Output)",
          "value": 30,
          "unit": "usd",
          "display": "$30.00",
          "context": "GPT-5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-5-5",
          "series": "Cache read",
          "point": "OpenAI",
          "metric": "provider-prices-gpt-5-5/Cache read",
          "label": "GPT-5.5: price per million tokens by provider (Cache read)",
          "value": 0.5,
          "unit": "usd",
          "display": "$0.50",
          "context": "GPT-5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "google-ai-studio",
      "name": "Google AI Studio",
      "vendor": "Google",
      "kind": "provider",
      "description": "Google’s Gemini API (AI Studio): the first-party list price of Gemini models, and its endpoints as listed on OpenRouter.",
      "aliases": [
        "Google AI Studio"
      ],
      "facts": [
        {
          "studySlug": "inference-provider-index",
          "chartId": "gateway-vs-direct-gemini-3-8-flash",
          "series": "Input",
          "point": "Google AI Studio (first-party list price)",
          "metric": "gateway-vs-direct-gemini-3-8-flash/Input",
          "label": "Gemini 3.8 Flash: OpenRouter vs Google list price (Input)",
          "value": 0.75,
          "unit": "usd",
          "display": "$0.75",
          "context": "first-party list price · Gemini 3.8 Flash · list price, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "gateway-vs-direct-gemini-3-8-flash",
          "series": "Output",
          "point": "Google AI Studio (first-party list price)",
          "metric": "gateway-vs-direct-gemini-3-8-flash/Output",
          "label": "Gemini 3.8 Flash: OpenRouter vs Google list price (Output)",
          "value": 3.75,
          "unit": "usd",
          "display": "$3.75",
          "context": "first-party list price · Gemini 3.8 Flash · list price, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "gateway-vs-direct-gemini-3-5-flash",
          "series": "Input",
          "point": "Google AI Studio (first-party list price)",
          "metric": "gateway-vs-direct-gemini-3-5-flash/Input",
          "label": "Gemini 3.5 Flash: OpenRouter vs Google list price (Input)",
          "value": 1.5,
          "unit": "usd",
          "display": "$1.50",
          "context": "first-party list price · Gemini 3.5 Flash · list price, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "gateway-vs-direct-gemini-3-5-flash",
          "series": "Output",
          "point": "Google AI Studio (first-party list price)",
          "metric": "gateway-vs-direct-gemini-3-5-flash/Output",
          "label": "Gemini 3.5 Flash: OpenRouter vs Google list price (Output)",
          "value": 9,
          "unit": "usd",
          "display": "$9.00",
          "context": "first-party list price · Gemini 3.5 Flash · list price, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gemini-3-8-flash",
          "series": "Input",
          "point": "Google AI Studio",
          "metric": "provider-prices-gemini-3-8-flash/Input",
          "label": "Gemini 3.8 Flash: price per million tokens by provider (Input)",
          "value": 0.75,
          "unit": "usd",
          "display": "$0.75",
          "context": "Gemini 3.8 Flash · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gemini-3-8-flash",
          "series": "Output",
          "point": "Google AI Studio",
          "metric": "provider-prices-gemini-3-8-flash/Output",
          "label": "Gemini 3.8 Flash: price per million tokens by provider (Output)",
          "value": 3.75,
          "unit": "usd",
          "display": "$3.75",
          "context": "Gemini 3.8 Flash · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gemini-3-8-flash",
          "series": "Cache read",
          "point": "Google AI Studio",
          "metric": "provider-prices-gemini-3-8-flash/Cache read",
          "label": "Gemini 3.8 Flash: price per million tokens by provider (Cache read)",
          "value": 0.075,
          "unit": "usd",
          "display": "$0.075",
          "context": "Gemini 3.8 Flash · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gemini-3-5-flash",
          "series": "Input",
          "point": "Google AI Studio",
          "metric": "provider-prices-gemini-3-5-flash/Input",
          "label": "Gemini 3.5 Flash: price per million tokens by provider (Input)",
          "value": 1.5,
          "unit": "usd",
          "display": "$1.50",
          "context": "Gemini 3.5 Flash · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gemini-3-5-flash",
          "series": "Output",
          "point": "Google AI Studio",
          "metric": "provider-prices-gemini-3-5-flash/Output",
          "label": "Gemini 3.5 Flash: price per million tokens by provider (Output)",
          "value": 9,
          "unit": "usd",
          "display": "$9.00",
          "context": "Gemini 3.5 Flash · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gemini-3-5-flash",
          "series": "Cache read",
          "point": "Google AI Studio",
          "metric": "provider-prices-gemini-3-5-flash/Cache read",
          "label": "Gemini 3.5 Flash: price per million tokens by provider (Cache read)",
          "value": 0.15,
          "unit": "usd",
          "display": "$0.15",
          "context": "Gemini 3.5 Flash · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gemini-3-5-flash-lite",
          "series": "Input",
          "point": "Google AI Studio",
          "metric": "provider-prices-gemini-3-5-flash-lite/Input",
          "label": "Gemini 3.5 Flash Lite: price per million tokens by provider (Input)",
          "value": 0.3,
          "unit": "usd",
          "display": "$0.30",
          "context": "Gemini 3.5 Flash Lite · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gemini-3-5-flash-lite",
          "series": "Output",
          "point": "Google AI Studio",
          "metric": "provider-prices-gemini-3-5-flash-lite/Output",
          "label": "Gemini 3.5 Flash Lite: price per million tokens by provider (Output)",
          "value": 2.5,
          "unit": "usd",
          "display": "$2.50",
          "context": "Gemini 3.5 Flash Lite · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gemini-3-5-flash-lite",
          "series": "Cache read",
          "point": "Google AI Studio",
          "metric": "provider-prices-gemini-3-5-flash-lite/Cache read",
          "label": "Gemini 3.5 Flash Lite: price per million tokens by provider (Cache read)",
          "value": 0.03,
          "unit": "usd",
          "display": "$0.030",
          "context": "Gemini 3.5 Flash Lite · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gemini-3-1-pro-preview",
          "series": "Input",
          "point": "Google AI Studio",
          "metric": "provider-prices-gemini-3-1-pro-preview/Input",
          "label": "Gemini 3.1 Pro Preview: price per million tokens by provider (Input)",
          "value": 2,
          "unit": "usd",
          "display": "$2.00",
          "context": "Gemini 3.1 Pro Preview · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gemini-3-1-pro-preview",
          "series": "Output",
          "point": "Google AI Studio",
          "metric": "provider-prices-gemini-3-1-pro-preview/Output",
          "label": "Gemini 3.1 Pro Preview: price per million tokens by provider (Output)",
          "value": 12,
          "unit": "usd",
          "display": "$12.00",
          "context": "Gemini 3.1 Pro Preview · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gemini-3-1-pro-preview",
          "series": "Cache read",
          "point": "Google AI Studio",
          "metric": "provider-prices-gemini-3-1-pro-preview/Cache read",
          "label": "Gemini 3.1 Pro Preview: price per million tokens by provider (Cache read)",
          "value": 0.2,
          "unit": "usd",
          "display": "$0.20",
          "context": "Gemini 3.1 Pro Preview · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "google-vertex",
      "name": "Google Vertex AI",
      "vendor": "Google",
      "kind": "provider",
      "description": "Google Cloud’s model platform. An inference provider listed on OpenRouter. Its prices here are the ones OpenRouter’s public API reported for its endpoints.",
      "aliases": [
        "Google Vertex"
      ],
      "facts": [
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-haiku-4-5",
          "series": "Input",
          "point": "Google Vertex",
          "metric": "provider-prices-claude-haiku-4-5/Input",
          "label": "Claude Haiku 4.5: price per million tokens by provider (Input)",
          "value": 1,
          "unit": "usd",
          "display": "$1.00",
          "context": "Claude Haiku 4.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-haiku-4-5",
          "series": "Output",
          "point": "Google Vertex",
          "metric": "provider-prices-claude-haiku-4-5/Output",
          "label": "Claude Haiku 4.5: price per million tokens by provider (Output)",
          "value": 5,
          "unit": "usd",
          "display": "$5.00",
          "context": "Claude Haiku 4.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-haiku-4-5",
          "series": "Cache read",
          "point": "Google Vertex",
          "metric": "provider-prices-claude-haiku-4-5/Cache read",
          "label": "Claude Haiku 4.5: price per million tokens by provider (Cache read)",
          "value": 0.1,
          "unit": "usd",
          "display": "$0.10",
          "context": "Claude Haiku 4.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-sonnet-5",
          "series": "Input",
          "point": "Google Vertex",
          "metric": "provider-prices-claude-sonnet-5/Input",
          "label": "Claude Sonnet 5: price per million tokens by provider (Input)",
          "value": 2,
          "unit": "usd",
          "display": "$2.00",
          "context": "Claude Sonnet 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-sonnet-5",
          "series": "Output",
          "point": "Google Vertex",
          "metric": "provider-prices-claude-sonnet-5/Output",
          "label": "Claude Sonnet 5: price per million tokens by provider (Output)",
          "value": 10,
          "unit": "usd",
          "display": "$10.00",
          "context": "Claude Sonnet 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-sonnet-5",
          "series": "Cache read",
          "point": "Google Vertex",
          "metric": "provider-prices-claude-sonnet-5/Cache read",
          "label": "Claude Sonnet 5: price per million tokens by provider (Cache read)",
          "value": 0.2,
          "unit": "usd",
          "display": "$0.20",
          "context": "Claude Sonnet 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-sonnet-5-5",
          "series": "Input",
          "point": "Google Vertex",
          "metric": "provider-prices-claude-sonnet-5-5/Input",
          "label": "Claude Sonnet 5.5: price per million tokens by provider (Input)",
          "value": 2,
          "unit": "usd",
          "display": "$2.00",
          "context": "Claude Sonnet 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-sonnet-5-5",
          "series": "Output",
          "point": "Google Vertex",
          "metric": "provider-prices-claude-sonnet-5-5/Output",
          "label": "Claude Sonnet 5.5: price per million tokens by provider (Output)",
          "value": 10,
          "unit": "usd",
          "display": "$10.00",
          "context": "Claude Sonnet 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-sonnet-5-5",
          "series": "Cache read",
          "point": "Google Vertex",
          "metric": "provider-prices-claude-sonnet-5-5/Cache read",
          "label": "Claude Sonnet 5.5: price per million tokens by provider (Cache read)",
          "value": 0.2,
          "unit": "usd",
          "display": "$0.20",
          "context": "Claude Sonnet 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-4-8",
          "series": "Input",
          "point": "Google Vertex",
          "metric": "provider-prices-claude-opus-4-8/Input",
          "label": "Claude Opus 4.8: price per million tokens by provider (Input)",
          "value": 5,
          "unit": "usd",
          "display": "$5.00",
          "context": "Claude Opus 4.8 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-4-8",
          "series": "Output",
          "point": "Google Vertex",
          "metric": "provider-prices-claude-opus-4-8/Output",
          "label": "Claude Opus 4.8: price per million tokens by provider (Output)",
          "value": 25,
          "unit": "usd",
          "display": "$25.00",
          "context": "Claude Opus 4.8 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-4-8",
          "series": "Cache read",
          "point": "Google Vertex",
          "metric": "provider-prices-claude-opus-4-8/Cache read",
          "label": "Claude Opus 4.8: price per million tokens by provider (Cache read)",
          "value": 0.5,
          "unit": "usd",
          "display": "$0.50",
          "context": "Claude Opus 4.8 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-5",
          "series": "Input",
          "point": "Google Vertex",
          "metric": "provider-prices-claude-opus-5/Input",
          "label": "Claude Opus 5: price per million tokens by provider (Input)",
          "value": 5,
          "unit": "usd",
          "display": "$5.00",
          "context": "Claude Opus 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-5",
          "series": "Output",
          "point": "Google Vertex",
          "metric": "provider-prices-claude-opus-5/Output",
          "label": "Claude Opus 5: price per million tokens by provider (Output)",
          "value": 25,
          "unit": "usd",
          "display": "$25.00",
          "context": "Claude Opus 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-5",
          "series": "Cache read",
          "point": "Google Vertex",
          "metric": "provider-prices-claude-opus-5/Cache read",
          "label": "Claude Opus 5: price per million tokens by provider (Cache read)",
          "value": 0.5,
          "unit": "usd",
          "display": "$0.50",
          "context": "Claude Opus 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-5-5",
          "series": "Input",
          "point": "Google Vertex",
          "metric": "provider-prices-claude-opus-5-5/Input",
          "label": "Claude Opus 5.5: price per million tokens by provider (Input)",
          "value": 4,
          "unit": "usd",
          "display": "$4.00",
          "context": "Claude Opus 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-5-5",
          "series": "Output",
          "point": "Google Vertex",
          "metric": "provider-prices-claude-opus-5-5/Output",
          "label": "Claude Opus 5.5: price per million tokens by provider (Output)",
          "value": 20,
          "unit": "usd",
          "display": "$20.00",
          "context": "Claude Opus 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-5-5",
          "series": "Cache read",
          "point": "Google Vertex",
          "metric": "provider-prices-claude-opus-5-5/Cache read",
          "label": "Claude Opus 5.5: price per million tokens by provider (Cache read)",
          "value": 0.2,
          "unit": "usd",
          "display": "$0.20",
          "context": "Claude Opus 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-fable-5-1",
          "series": "Input",
          "point": "Google Vertex",
          "metric": "provider-prices-claude-fable-5-1/Input",
          "label": "Claude Fable 5.1: price per million tokens by provider (Input)",
          "value": 10,
          "unit": "usd",
          "display": "$10.00",
          "context": "Claude Fable 5.1 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-fable-5-1",
          "series": "Output",
          "point": "Google Vertex",
          "metric": "provider-prices-claude-fable-5-1/Output",
          "label": "Claude Fable 5.1: price per million tokens by provider (Output)",
          "value": 50,
          "unit": "usd",
          "display": "$50.00",
          "context": "Claude Fable 5.1 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-fable-5-1",
          "series": "Cache read",
          "point": "Google Vertex",
          "metric": "provider-prices-claude-fable-5-1/Cache read",
          "label": "Claude Fable 5.1: price per million tokens by provider (Cache read)",
          "value": 0.25,
          "unit": "usd",
          "display": "$0.25",
          "context": "Claude Fable 5.1 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "series": "Input",
          "point": "Google Vertex",
          "metric": "provider-prices-gpt-oss-120b/Input",
          "label": "gpt-oss-120b: price per million tokens by provider (Input)",
          "value": 0.09,
          "unit": "usd",
          "display": "$0.090",
          "context": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "series": "Output",
          "point": "Google Vertex",
          "metric": "provider-prices-gpt-oss-120b/Output",
          "label": "gpt-oss-120b: price per million tokens by provider (Output)",
          "value": 0.36,
          "unit": "usd",
          "display": "$0.36",
          "context": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gemini-3-8-flash",
          "series": "Input",
          "point": "Google Vertex",
          "metric": "provider-prices-gemini-3-8-flash/Input",
          "label": "Gemini 3.8 Flash: price per million tokens by provider (Input)",
          "value": 0.75,
          "unit": "usd",
          "display": "$0.75",
          "context": "Gemini 3.8 Flash · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gemini-3-8-flash",
          "series": "Output",
          "point": "Google Vertex",
          "metric": "provider-prices-gemini-3-8-flash/Output",
          "label": "Gemini 3.8 Flash: price per million tokens by provider (Output)",
          "value": 3.75,
          "unit": "usd",
          "display": "$3.75",
          "context": "Gemini 3.8 Flash · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gemini-3-8-flash",
          "series": "Cache read",
          "point": "Google Vertex",
          "metric": "provider-prices-gemini-3-8-flash/Cache read",
          "label": "Gemini 3.8 Flash: price per million tokens by provider (Cache read)",
          "value": 0.075,
          "unit": "usd",
          "display": "$0.075",
          "context": "Gemini 3.8 Flash · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gemini-3-5-flash",
          "series": "Input",
          "point": "Google Vertex",
          "metric": "provider-prices-gemini-3-5-flash/Input",
          "label": "Gemini 3.5 Flash: price per million tokens by provider (Input)",
          "value": 1.5,
          "unit": "usd",
          "display": "$1.50",
          "context": "Gemini 3.5 Flash · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gemini-3-5-flash",
          "series": "Output",
          "point": "Google Vertex",
          "metric": "provider-prices-gemini-3-5-flash/Output",
          "label": "Gemini 3.5 Flash: price per million tokens by provider (Output)",
          "value": 9,
          "unit": "usd",
          "display": "$9.00",
          "context": "Gemini 3.5 Flash · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gemini-3-5-flash",
          "series": "Cache read",
          "point": "Google Vertex",
          "metric": "provider-prices-gemini-3-5-flash/Cache read",
          "label": "Gemini 3.5 Flash: price per million tokens by provider (Cache read)",
          "value": 0.15,
          "unit": "usd",
          "display": "$0.15",
          "context": "Gemini 3.5 Flash · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gemini-3-5-flash-lite",
          "series": "Input",
          "point": "Google Vertex",
          "metric": "provider-prices-gemini-3-5-flash-lite/Input",
          "label": "Gemini 3.5 Flash Lite: price per million tokens by provider (Input)",
          "value": 0.3,
          "unit": "usd",
          "display": "$0.30",
          "context": "Gemini 3.5 Flash Lite · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gemini-3-5-flash-lite",
          "series": "Output",
          "point": "Google Vertex",
          "metric": "provider-prices-gemini-3-5-flash-lite/Output",
          "label": "Gemini 3.5 Flash Lite: price per million tokens by provider (Output)",
          "value": 2.5,
          "unit": "usd",
          "display": "$2.50",
          "context": "Gemini 3.5 Flash Lite · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gemini-3-5-flash-lite",
          "series": "Cache read",
          "point": "Google Vertex",
          "metric": "provider-prices-gemini-3-5-flash-lite/Cache read",
          "label": "Gemini 3.5 Flash Lite: price per million tokens by provider (Cache read)",
          "value": 0.03,
          "unit": "usd",
          "display": "$0.030",
          "context": "Gemini 3.5 Flash Lite · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gemini-3-1-pro-preview",
          "series": "Input",
          "point": "Google Vertex",
          "metric": "provider-prices-gemini-3-1-pro-preview/Input",
          "label": "Gemini 3.1 Pro Preview: price per million tokens by provider (Input)",
          "value": 2,
          "unit": "usd",
          "display": "$2.00",
          "context": "Gemini 3.1 Pro Preview · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gemini-3-1-pro-preview",
          "series": "Output",
          "point": "Google Vertex",
          "metric": "provider-prices-gemini-3-1-pro-preview/Output",
          "label": "Gemini 3.1 Pro Preview: price per million tokens by provider (Output)",
          "value": 12,
          "unit": "usd",
          "display": "$12.00",
          "context": "Gemini 3.1 Pro Preview · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gemini-3-1-pro-preview",
          "series": "Cache read",
          "point": "Google Vertex",
          "metric": "provider-prices-gemini-3-1-pro-preview/Cache read",
          "label": "Gemini 3.1 Pro Preview: price per million tokens by provider (Cache read)",
          "value": 0.2,
          "unit": "usd",
          "display": "$0.20",
          "context": "Gemini 3.1 Pro Preview · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-llama-3-3-70b-instruct",
          "series": "Input",
          "point": "Google Vertex",
          "metric": "provider-prices-llama-3-3-70b-instruct/Input",
          "label": "Llama 3.3 70B Instruct: price per million tokens by provider (Input)",
          "value": 0.72,
          "unit": "usd",
          "display": "$0.72",
          "context": "Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-llama-3-3-70b-instruct",
          "series": "Output",
          "point": "Google Vertex",
          "metric": "provider-prices-llama-3-3-70b-instruct/Output",
          "label": "Llama 3.3 70B Instruct: price per million tokens by provider (Output)",
          "value": 0.72,
          "unit": "usd",
          "display": "$0.72",
          "context": "Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "amazon-bedrock",
      "name": "Amazon Bedrock",
      "vendor": "Amazon",
      "kind": "provider",
      "description": "Amazon Web Services’ model platform. An inference provider listed on OpenRouter. Its prices here are the ones OpenRouter’s public API reported for its endpoints.",
      "aliases": [
        "Amazon Bedrock"
      ],
      "facts": [
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-haiku-4-5",
          "series": "Input",
          "point": "Amazon Bedrock",
          "metric": "provider-prices-claude-haiku-4-5/Input",
          "label": "Claude Haiku 4.5: price per million tokens by provider (Input)",
          "value": 1,
          "unit": "usd",
          "display": "$1.00",
          "context": "Claude Haiku 4.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-haiku-4-5",
          "series": "Output",
          "point": "Amazon Bedrock",
          "metric": "provider-prices-claude-haiku-4-5/Output",
          "label": "Claude Haiku 4.5: price per million tokens by provider (Output)",
          "value": 5,
          "unit": "usd",
          "display": "$5.00",
          "context": "Claude Haiku 4.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-haiku-4-5",
          "series": "Cache read",
          "point": "Amazon Bedrock",
          "metric": "provider-prices-claude-haiku-4-5/Cache read",
          "label": "Claude Haiku 4.5: price per million tokens by provider (Cache read)",
          "value": 0.1,
          "unit": "usd",
          "display": "$0.10",
          "context": "Claude Haiku 4.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-sonnet-5",
          "series": "Input",
          "point": "Amazon Bedrock",
          "metric": "provider-prices-claude-sonnet-5/Input",
          "label": "Claude Sonnet 5: price per million tokens by provider (Input)",
          "value": 2,
          "unit": "usd",
          "display": "$2.00",
          "context": "Claude Sonnet 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-sonnet-5",
          "series": "Output",
          "point": "Amazon Bedrock",
          "metric": "provider-prices-claude-sonnet-5/Output",
          "label": "Claude Sonnet 5: price per million tokens by provider (Output)",
          "value": 10,
          "unit": "usd",
          "display": "$10.00",
          "context": "Claude Sonnet 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-sonnet-5",
          "series": "Cache read",
          "point": "Amazon Bedrock",
          "metric": "provider-prices-claude-sonnet-5/Cache read",
          "label": "Claude Sonnet 5: price per million tokens by provider (Cache read)",
          "value": 0.2,
          "unit": "usd",
          "display": "$0.20",
          "context": "Claude Sonnet 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-sonnet-5-5",
          "series": "Input",
          "point": "Amazon Bedrock",
          "metric": "provider-prices-claude-sonnet-5-5/Input",
          "label": "Claude Sonnet 5.5: price per million tokens by provider (Input)",
          "value": 2,
          "unit": "usd",
          "display": "$2.00",
          "context": "Claude Sonnet 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-sonnet-5-5",
          "series": "Output",
          "point": "Amazon Bedrock",
          "metric": "provider-prices-claude-sonnet-5-5/Output",
          "label": "Claude Sonnet 5.5: price per million tokens by provider (Output)",
          "value": 10,
          "unit": "usd",
          "display": "$10.00",
          "context": "Claude Sonnet 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-sonnet-5-5",
          "series": "Cache read",
          "point": "Amazon Bedrock",
          "metric": "provider-prices-claude-sonnet-5-5/Cache read",
          "label": "Claude Sonnet 5.5: price per million tokens by provider (Cache read)",
          "value": 0.2,
          "unit": "usd",
          "display": "$0.20",
          "context": "Claude Sonnet 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-4-8",
          "series": "Input",
          "point": "Amazon Bedrock",
          "metric": "provider-prices-claude-opus-4-8/Input",
          "label": "Claude Opus 4.8: price per million tokens by provider (Input)",
          "value": 5,
          "unit": "usd",
          "display": "$5.00",
          "context": "Claude Opus 4.8 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-4-8",
          "series": "Output",
          "point": "Amazon Bedrock",
          "metric": "provider-prices-claude-opus-4-8/Output",
          "label": "Claude Opus 4.8: price per million tokens by provider (Output)",
          "value": 25,
          "unit": "usd",
          "display": "$25.00",
          "context": "Claude Opus 4.8 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-4-8",
          "series": "Cache read",
          "point": "Amazon Bedrock",
          "metric": "provider-prices-claude-opus-4-8/Cache read",
          "label": "Claude Opus 4.8: price per million tokens by provider (Cache read)",
          "value": 0.5,
          "unit": "usd",
          "display": "$0.50",
          "context": "Claude Opus 4.8 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-5",
          "series": "Input",
          "point": "Amazon Bedrock",
          "metric": "provider-prices-claude-opus-5/Input",
          "label": "Claude Opus 5: price per million tokens by provider (Input)",
          "value": 5,
          "unit": "usd",
          "display": "$5.00",
          "context": "Claude Opus 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-5",
          "series": "Output",
          "point": "Amazon Bedrock",
          "metric": "provider-prices-claude-opus-5/Output",
          "label": "Claude Opus 5: price per million tokens by provider (Output)",
          "value": 25,
          "unit": "usd",
          "display": "$25.00",
          "context": "Claude Opus 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-5",
          "series": "Cache read",
          "point": "Amazon Bedrock",
          "metric": "provider-prices-claude-opus-5/Cache read",
          "label": "Claude Opus 5: price per million tokens by provider (Cache read)",
          "value": 0.5,
          "unit": "usd",
          "display": "$0.50",
          "context": "Claude Opus 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-5-5",
          "series": "Input",
          "point": "Amazon Bedrock",
          "metric": "provider-prices-claude-opus-5-5/Input",
          "label": "Claude Opus 5.5: price per million tokens by provider (Input)",
          "value": 4,
          "unit": "usd",
          "display": "$4.00",
          "context": "Claude Opus 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-5-5",
          "series": "Output",
          "point": "Amazon Bedrock",
          "metric": "provider-prices-claude-opus-5-5/Output",
          "label": "Claude Opus 5.5: price per million tokens by provider (Output)",
          "value": 20,
          "unit": "usd",
          "display": "$20.00",
          "context": "Claude Opus 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-5-5",
          "series": "Cache read",
          "point": "Amazon Bedrock",
          "metric": "provider-prices-claude-opus-5-5/Cache read",
          "label": "Claude Opus 5.5: price per million tokens by provider (Cache read)",
          "value": 0.2,
          "unit": "usd",
          "display": "$0.20",
          "context": "Claude Opus 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-fable-5-1",
          "series": "Input",
          "point": "Amazon Bedrock",
          "metric": "provider-prices-claude-fable-5-1/Input",
          "label": "Claude Fable 5.1: price per million tokens by provider (Input)",
          "value": 10,
          "unit": "usd",
          "display": "$10.00",
          "context": "Claude Fable 5.1 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-fable-5-1",
          "series": "Output",
          "point": "Amazon Bedrock",
          "metric": "provider-prices-claude-fable-5-1/Output",
          "label": "Claude Fable 5.1: price per million tokens by provider (Output)",
          "value": 50,
          "unit": "usd",
          "display": "$50.00",
          "context": "Claude Fable 5.1 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-fable-5-1",
          "series": "Cache read",
          "point": "Amazon Bedrock",
          "metric": "provider-prices-claude-fable-5-1/Cache read",
          "label": "Claude Fable 5.1: price per million tokens by provider (Cache read)",
          "value": 0.25,
          "unit": "usd",
          "display": "$0.25",
          "context": "Claude Fable 5.1 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "series": "Input",
          "point": "Amazon Bedrock",
          "metric": "provider-prices-gpt-oss-120b/Input",
          "label": "gpt-oss-120b: price per million tokens by provider (Input)",
          "value": 0.15,
          "unit": "usd",
          "display": "$0.15",
          "context": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "series": "Output",
          "point": "Amazon Bedrock",
          "metric": "provider-prices-gpt-oss-120b/Output",
          "label": "gpt-oss-120b: price per million tokens by provider (Output)",
          "value": 0.6,
          "unit": "usd",
          "display": "$0.60",
          "context": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "azure",
      "name": "Azure",
      "vendor": "Microsoft",
      "kind": "provider",
      "description": "Microsoft’s cloud model platform. An inference provider listed on OpenRouter. Its prices here are the ones OpenRouter’s public API reported for its endpoints.",
      "aliases": [
        "Azure"
      ],
      "facts": [
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-haiku-4-5",
          "series": "Input",
          "point": "Azure",
          "metric": "provider-prices-claude-haiku-4-5/Input",
          "label": "Claude Haiku 4.5: price per million tokens by provider (Input)",
          "value": 1,
          "unit": "usd",
          "display": "$1.00",
          "context": "Claude Haiku 4.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-haiku-4-5",
          "series": "Output",
          "point": "Azure",
          "metric": "provider-prices-claude-haiku-4-5/Output",
          "label": "Claude Haiku 4.5: price per million tokens by provider (Output)",
          "value": 5,
          "unit": "usd",
          "display": "$5.00",
          "context": "Claude Haiku 4.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-haiku-4-5",
          "series": "Cache read",
          "point": "Azure",
          "metric": "provider-prices-claude-haiku-4-5/Cache read",
          "label": "Claude Haiku 4.5: price per million tokens by provider (Cache read)",
          "value": 0.1,
          "unit": "usd",
          "display": "$0.10",
          "context": "Claude Haiku 4.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-sonnet-5",
          "series": "Input",
          "point": "Azure",
          "metric": "provider-prices-claude-sonnet-5/Input",
          "label": "Claude Sonnet 5: price per million tokens by provider (Input)",
          "value": 2,
          "unit": "usd",
          "display": "$2.00",
          "context": "Claude Sonnet 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-sonnet-5",
          "series": "Output",
          "point": "Azure",
          "metric": "provider-prices-claude-sonnet-5/Output",
          "label": "Claude Sonnet 5: price per million tokens by provider (Output)",
          "value": 10,
          "unit": "usd",
          "display": "$10.00",
          "context": "Claude Sonnet 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-sonnet-5",
          "series": "Cache read",
          "point": "Azure",
          "metric": "provider-prices-claude-sonnet-5/Cache read",
          "label": "Claude Sonnet 5: price per million tokens by provider (Cache read)",
          "value": 0.2,
          "unit": "usd",
          "display": "$0.20",
          "context": "Claude Sonnet 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-sonnet-5-5",
          "series": "Input",
          "point": "Azure",
          "metric": "provider-prices-claude-sonnet-5-5/Input",
          "label": "Claude Sonnet 5.5: price per million tokens by provider (Input)",
          "value": 2,
          "unit": "usd",
          "display": "$2.00",
          "context": "Claude Sonnet 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-sonnet-5-5",
          "series": "Output",
          "point": "Azure",
          "metric": "provider-prices-claude-sonnet-5-5/Output",
          "label": "Claude Sonnet 5.5: price per million tokens by provider (Output)",
          "value": 10,
          "unit": "usd",
          "display": "$10.00",
          "context": "Claude Sonnet 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-sonnet-5-5",
          "series": "Cache read",
          "point": "Azure",
          "metric": "provider-prices-claude-sonnet-5-5/Cache read",
          "label": "Claude Sonnet 5.5: price per million tokens by provider (Cache read)",
          "value": 0.2,
          "unit": "usd",
          "display": "$0.20",
          "context": "Claude Sonnet 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-4-8",
          "series": "Input",
          "point": "Azure",
          "metric": "provider-prices-claude-opus-4-8/Input",
          "label": "Claude Opus 4.8: price per million tokens by provider (Input)",
          "value": 5,
          "unit": "usd",
          "display": "$5.00",
          "context": "Claude Opus 4.8 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-4-8",
          "series": "Output",
          "point": "Azure",
          "metric": "provider-prices-claude-opus-4-8/Output",
          "label": "Claude Opus 4.8: price per million tokens by provider (Output)",
          "value": 25,
          "unit": "usd",
          "display": "$25.00",
          "context": "Claude Opus 4.8 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-4-8",
          "series": "Cache read",
          "point": "Azure",
          "metric": "provider-prices-claude-opus-4-8/Cache read",
          "label": "Claude Opus 4.8: price per million tokens by provider (Cache read)",
          "value": 0.5,
          "unit": "usd",
          "display": "$0.50",
          "context": "Claude Opus 4.8 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-5",
          "series": "Input",
          "point": "Azure",
          "metric": "provider-prices-claude-opus-5/Input",
          "label": "Claude Opus 5: price per million tokens by provider (Input)",
          "value": 5,
          "unit": "usd",
          "display": "$5.00",
          "context": "Claude Opus 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-5",
          "series": "Output",
          "point": "Azure",
          "metric": "provider-prices-claude-opus-5/Output",
          "label": "Claude Opus 5: price per million tokens by provider (Output)",
          "value": 25,
          "unit": "usd",
          "display": "$25.00",
          "context": "Claude Opus 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-5",
          "series": "Cache read",
          "point": "Azure",
          "metric": "provider-prices-claude-opus-5/Cache read",
          "label": "Claude Opus 5: price per million tokens by provider (Cache read)",
          "value": 0.5,
          "unit": "usd",
          "display": "$0.50",
          "context": "Claude Opus 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-5-5",
          "series": "Input",
          "point": "Azure",
          "metric": "provider-prices-claude-opus-5-5/Input",
          "label": "Claude Opus 5.5: price per million tokens by provider (Input)",
          "value": 4,
          "unit": "usd",
          "display": "$4.00",
          "context": "Claude Opus 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-5-5",
          "series": "Output",
          "point": "Azure",
          "metric": "provider-prices-claude-opus-5-5/Output",
          "label": "Claude Opus 5.5: price per million tokens by provider (Output)",
          "value": 20,
          "unit": "usd",
          "display": "$20.00",
          "context": "Claude Opus 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-5-5",
          "series": "Cache read",
          "point": "Azure",
          "metric": "provider-prices-claude-opus-5-5/Cache read",
          "label": "Claude Opus 5.5: price per million tokens by provider (Cache read)",
          "value": 0.2,
          "unit": "usd",
          "display": "$0.20",
          "context": "Claude Opus 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-fable-5-1",
          "series": "Input",
          "point": "Azure",
          "metric": "provider-prices-claude-fable-5-1/Input",
          "label": "Claude Fable 5.1: price per million tokens by provider (Input)",
          "value": 10,
          "unit": "usd",
          "display": "$10.00",
          "context": "Claude Fable 5.1 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-fable-5-1",
          "series": "Output",
          "point": "Azure",
          "metric": "provider-prices-claude-fable-5-1/Output",
          "label": "Claude Fable 5.1: price per million tokens by provider (Output)",
          "value": 50,
          "unit": "usd",
          "display": "$50.00",
          "context": "Claude Fable 5.1 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-fable-5-1",
          "series": "Cache read",
          "point": "Azure",
          "metric": "provider-prices-claude-fable-5-1/Cache read",
          "label": "Claude Fable 5.1: price per million tokens by provider (Cache read)",
          "value": 0.25,
          "unit": "usd",
          "display": "$0.25",
          "context": "Claude Fable 5.1 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-6-sol",
          "series": "Input",
          "point": "Azure",
          "metric": "provider-prices-gpt-6-sol/Input",
          "label": "GPT-6 Sol: price per million tokens by provider (Input)",
          "value": 2,
          "unit": "usd",
          "display": "$2.00",
          "context": "GPT-6 Sol · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-6-sol",
          "series": "Output",
          "point": "Azure",
          "metric": "provider-prices-gpt-6-sol/Output",
          "label": "GPT-6 Sol: price per million tokens by provider (Output)",
          "value": 10,
          "unit": "usd",
          "display": "$10.00",
          "context": "GPT-6 Sol · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-6-sol",
          "series": "Cache read",
          "point": "Azure",
          "metric": "provider-prices-gpt-6-sol/Cache read",
          "label": "GPT-6 Sol: price per million tokens by provider (Cache read)",
          "value": 0.2,
          "unit": "usd",
          "display": "$0.20",
          "context": "GPT-6 Sol · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-6-luna",
          "series": "Input",
          "point": "Azure",
          "metric": "provider-prices-gpt-6-luna/Input",
          "label": "GPT-6 Luna: price per million tokens by provider (Input)",
          "value": 0.1,
          "unit": "usd",
          "display": "$0.10",
          "context": "GPT-6 Luna · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-6-luna",
          "series": "Output",
          "point": "Azure",
          "metric": "provider-prices-gpt-6-luna/Output",
          "label": "GPT-6 Luna: price per million tokens by provider (Output)",
          "value": 0.5,
          "unit": "usd",
          "display": "$0.50",
          "context": "GPT-6 Luna · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-6-luna",
          "series": "Cache read",
          "point": "Azure",
          "metric": "provider-prices-gpt-6-luna/Cache read",
          "label": "GPT-6 Luna: price per million tokens by provider (Cache read)",
          "value": 0.01,
          "unit": "usd",
          "display": "$0.010",
          "context": "GPT-6 Luna · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-6-astra",
          "series": "Input",
          "point": "Azure",
          "metric": "provider-prices-gpt-6-astra/Input",
          "label": "GPT-6 Astra: price per million tokens by provider (Input)",
          "value": 10,
          "unit": "usd",
          "display": "$10.00",
          "context": "GPT-6 Astra · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-6-astra",
          "series": "Output",
          "point": "Azure",
          "metric": "provider-prices-gpt-6-astra/Output",
          "label": "GPT-6 Astra: price per million tokens by provider (Output)",
          "value": 50,
          "unit": "usd",
          "display": "$50.00",
          "context": "GPT-6 Astra · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-6-astra",
          "series": "Cache read",
          "point": "Azure",
          "metric": "provider-prices-gpt-6-astra/Cache read",
          "label": "GPT-6 Astra: price per million tokens by provider (Cache read)",
          "value": 1,
          "unit": "usd",
          "display": "$1.00",
          "context": "GPT-6 Astra · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-5-5",
          "series": "Input",
          "point": "Azure",
          "metric": "provider-prices-gpt-5-5/Input",
          "label": "GPT-5.5: price per million tokens by provider (Input)",
          "value": 5,
          "unit": "usd",
          "display": "$5.00",
          "context": "GPT-5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-5-5",
          "series": "Output",
          "point": "Azure",
          "metric": "provider-prices-gpt-5-5/Output",
          "label": "GPT-5.5: price per million tokens by provider (Output)",
          "value": 30,
          "unit": "usd",
          "display": "$30.00",
          "context": "GPT-5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-5-5",
          "series": "Cache read",
          "point": "Azure",
          "metric": "provider-prices-gpt-5-5/Cache read",
          "label": "GPT-5.5: price per million tokens by provider (Cache read)",
          "value": 0.5,
          "unit": "usd",
          "display": "$0.50",
          "context": "GPT-5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "claude-platform-on-aws",
      "name": "Claude Platform on AWS",
      "vendor": "Anthropic",
      "kind": "provider",
      "description": "Anthropic’s Claude platform hosted on AWS. An inference provider listed on OpenRouter. Its prices here are the ones OpenRouter’s public API reported for its endpoints.",
      "aliases": [
        "Claude Platform on AWS"
      ],
      "facts": [
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-sonnet-5",
          "series": "Input",
          "point": "Claude Platform on AWS",
          "metric": "provider-prices-claude-sonnet-5/Input",
          "label": "Claude Sonnet 5: price per million tokens by provider (Input)",
          "value": 2,
          "unit": "usd",
          "display": "$2.00",
          "context": "Claude Sonnet 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-sonnet-5",
          "series": "Output",
          "point": "Claude Platform on AWS",
          "metric": "provider-prices-claude-sonnet-5/Output",
          "label": "Claude Sonnet 5: price per million tokens by provider (Output)",
          "value": 10,
          "unit": "usd",
          "display": "$10.00",
          "context": "Claude Sonnet 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-sonnet-5",
          "series": "Cache read",
          "point": "Claude Platform on AWS",
          "metric": "provider-prices-claude-sonnet-5/Cache read",
          "label": "Claude Sonnet 5: price per million tokens by provider (Cache read)",
          "value": 0.2,
          "unit": "usd",
          "display": "$0.20",
          "context": "Claude Sonnet 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-sonnet-5-5",
          "series": "Input",
          "point": "Claude Platform on AWS",
          "metric": "provider-prices-claude-sonnet-5-5/Input",
          "label": "Claude Sonnet 5.5: price per million tokens by provider (Input)",
          "value": 2,
          "unit": "usd",
          "display": "$2.00",
          "context": "Claude Sonnet 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-sonnet-5-5",
          "series": "Output",
          "point": "Claude Platform on AWS",
          "metric": "provider-prices-claude-sonnet-5-5/Output",
          "label": "Claude Sonnet 5.5: price per million tokens by provider (Output)",
          "value": 10,
          "unit": "usd",
          "display": "$10.00",
          "context": "Claude Sonnet 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-sonnet-5-5",
          "series": "Cache read",
          "point": "Claude Platform on AWS",
          "metric": "provider-prices-claude-sonnet-5-5/Cache read",
          "label": "Claude Sonnet 5.5: price per million tokens by provider (Cache read)",
          "value": 0.2,
          "unit": "usd",
          "display": "$0.20",
          "context": "Claude Sonnet 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-4-8",
          "series": "Input",
          "point": "Claude Platform on AWS",
          "metric": "provider-prices-claude-opus-4-8/Input",
          "label": "Claude Opus 4.8: price per million tokens by provider (Input)",
          "value": 5,
          "unit": "usd",
          "display": "$5.00",
          "context": "Claude Opus 4.8 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-4-8",
          "series": "Output",
          "point": "Claude Platform on AWS",
          "metric": "provider-prices-claude-opus-4-8/Output",
          "label": "Claude Opus 4.8: price per million tokens by provider (Output)",
          "value": 25,
          "unit": "usd",
          "display": "$25.00",
          "context": "Claude Opus 4.8 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-4-8",
          "series": "Cache read",
          "point": "Claude Platform on AWS",
          "metric": "provider-prices-claude-opus-4-8/Cache read",
          "label": "Claude Opus 4.8: price per million tokens by provider (Cache read)",
          "value": 0.5,
          "unit": "usd",
          "display": "$0.50",
          "context": "Claude Opus 4.8 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-5",
          "series": "Input",
          "point": "Claude Platform on AWS",
          "metric": "provider-prices-claude-opus-5/Input",
          "label": "Claude Opus 5: price per million tokens by provider (Input)",
          "value": 5,
          "unit": "usd",
          "display": "$5.00",
          "context": "Claude Opus 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-5",
          "series": "Output",
          "point": "Claude Platform on AWS",
          "metric": "provider-prices-claude-opus-5/Output",
          "label": "Claude Opus 5: price per million tokens by provider (Output)",
          "value": 25,
          "unit": "usd",
          "display": "$25.00",
          "context": "Claude Opus 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-5",
          "series": "Cache read",
          "point": "Claude Platform on AWS",
          "metric": "provider-prices-claude-opus-5/Cache read",
          "label": "Claude Opus 5: price per million tokens by provider (Cache read)",
          "value": 0.5,
          "unit": "usd",
          "display": "$0.50",
          "context": "Claude Opus 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-5-5",
          "series": "Input",
          "point": "Claude Platform on AWS",
          "metric": "provider-prices-claude-opus-5-5/Input",
          "label": "Claude Opus 5.5: price per million tokens by provider (Input)",
          "value": 4,
          "unit": "usd",
          "display": "$4.00",
          "context": "Claude Opus 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-5-5",
          "series": "Output",
          "point": "Claude Platform on AWS",
          "metric": "provider-prices-claude-opus-5-5/Output",
          "label": "Claude Opus 5.5: price per million tokens by provider (Output)",
          "value": 20,
          "unit": "usd",
          "display": "$20.00",
          "context": "Claude Opus 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-5-5",
          "series": "Cache read",
          "point": "Claude Platform on AWS",
          "metric": "provider-prices-claude-opus-5-5/Cache read",
          "label": "Claude Opus 5.5: price per million tokens by provider (Cache read)",
          "value": 0.2,
          "unit": "usd",
          "display": "$0.20",
          "context": "Claude Opus 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "groq",
      "name": "Groq",
      "vendor": "Groq",
      "kind": "provider",
      "description": "An inference provider listed on OpenRouter. Its prices here are the ones OpenRouter’s public API reported for its endpoints.",
      "aliases": [
        "Groq"
      ],
      "facts": [
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "series": "Input",
          "point": "Groq",
          "metric": "provider-prices-gpt-oss-120b/Input",
          "label": "gpt-oss-120b: price per million tokens by provider (Input)",
          "value": 0.15,
          "unit": "usd",
          "display": "$0.15",
          "context": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "series": "Output",
          "point": "Groq",
          "metric": "provider-prices-gpt-oss-120b/Output",
          "label": "gpt-oss-120b: price per million tokens by provider (Output)",
          "value": 0.6,
          "unit": "usd",
          "display": "$0.60",
          "context": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "series": "Cache read",
          "point": "Groq",
          "metric": "provider-prices-gpt-oss-120b/Cache read",
          "label": "gpt-oss-120b: price per million tokens by provider (Cache read)",
          "value": 0.075,
          "unit": "usd",
          "display": "$0.075",
          "context": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-llama-3-3-70b-instruct",
          "series": "Input",
          "point": "Groq",
          "metric": "provider-prices-llama-3-3-70b-instruct/Input",
          "label": "Llama 3.3 70B Instruct: price per million tokens by provider (Input)",
          "value": 0.59,
          "unit": "usd",
          "display": "$0.59",
          "context": "Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-llama-3-3-70b-instruct",
          "series": "Output",
          "point": "Groq",
          "metric": "provider-prices-llama-3-3-70b-instruct/Output",
          "label": "Llama 3.3 70B Instruct: price per million tokens by provider (Output)",
          "value": 0.79,
          "unit": "usd",
          "display": "$0.79",
          "context": "Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-llama-3-3-70b-instruct",
          "series": "Cache read",
          "point": "Groq",
          "metric": "provider-prices-llama-3-3-70b-instruct/Cache read",
          "label": "Llama 3.3 70B Instruct: price per million tokens by provider (Cache read)",
          "value": 0.295,
          "unit": "usd",
          "display": "$0.29",
          "context": "Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "together",
      "name": "Together AI",
      "vendor": "Together AI",
      "kind": "provider",
      "description": "An inference provider listed on OpenRouter. Its prices here are the ones OpenRouter’s public API reported for its endpoints.",
      "aliases": [
        "Together"
      ],
      "facts": [
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "series": "Input",
          "point": "Together",
          "metric": "provider-prices-gpt-oss-120b/Input",
          "label": "gpt-oss-120b: price per million tokens by provider (Input)",
          "value": 0.15,
          "unit": "usd",
          "display": "$0.15",
          "context": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "series": "Output",
          "point": "Together",
          "metric": "provider-prices-gpt-oss-120b/Output",
          "label": "gpt-oss-120b: price per million tokens by provider (Output)",
          "value": 0.6,
          "unit": "usd",
          "display": "$0.60",
          "context": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-llama-3-3-70b-instruct",
          "series": "Input",
          "point": "Together",
          "metric": "provider-prices-llama-3-3-70b-instruct/Input",
          "label": "Llama 3.3 70B Instruct: price per million tokens by provider (Input)",
          "value": 1.04,
          "unit": "usd",
          "display": "$1.04",
          "context": "Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-llama-3-3-70b-instruct",
          "series": "Output",
          "point": "Together",
          "metric": "provider-prices-llama-3-3-70b-instruct/Output",
          "label": "Llama 3.3 70B Instruct: price per million tokens by provider (Output)",
          "value": 1.04,
          "unit": "usd",
          "display": "$1.04",
          "context": "Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-kimi-k3",
          "series": "Input",
          "point": "Together",
          "metric": "provider-prices-kimi-k3/Input",
          "label": "Kimi K3: price per million tokens by provider (Input)",
          "value": 2.7,
          "unit": "usd",
          "display": "$2.70",
          "context": "Kimi K3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-kimi-k3",
          "series": "Output",
          "point": "Together",
          "metric": "provider-prices-kimi-k3/Output",
          "label": "Kimi K3: price per million tokens by provider (Output)",
          "value": 13.5,
          "unit": "usd",
          "display": "$13.50",
          "context": "Kimi K3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-kimi-k3",
          "series": "Cache read",
          "point": "Together",
          "metric": "provider-prices-kimi-k3/Cache read",
          "label": "Kimi K3: price per million tokens by provider (Cache read)",
          "value": 0.27,
          "unit": "usd",
          "display": "$0.27",
          "context": "Kimi K3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "series": "Input",
          "point": "Together",
          "metric": "provider-prices-glm-5-3/Input",
          "label": "GLM 5.3: price per million tokens by provider (Input)",
          "value": 1.4,
          "unit": "usd",
          "display": "$1.40",
          "context": "GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "series": "Output",
          "point": "Together",
          "metric": "provider-prices-glm-5-3/Output",
          "label": "GLM 5.3: price per million tokens by provider (Output)",
          "value": 4.4,
          "unit": "usd",
          "display": "$4.40",
          "context": "GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "series": "Cache read",
          "point": "Together",
          "metric": "provider-prices-glm-5-3/Cache read",
          "label": "GLM 5.3: price per million tokens by provider (Cache read)",
          "value": 0.26,
          "unit": "usd",
          "display": "$0.26",
          "context": "GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "fireworks",
      "name": "Fireworks AI",
      "vendor": "Fireworks AI",
      "kind": "provider",
      "description": "An inference provider listed on OpenRouter. Its prices here are the ones OpenRouter’s public API reported for its endpoints.",
      "aliases": [
        "Fireworks"
      ],
      "facts": [
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-kimi-k3",
          "series": "Input",
          "point": "Fireworks",
          "metric": "provider-prices-kimi-k3/Input",
          "label": "Kimi K3: price per million tokens by provider (Input)",
          "value": 3,
          "unit": "usd",
          "display": "$3.00",
          "context": "Kimi K3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-kimi-k3",
          "series": "Output",
          "point": "Fireworks",
          "metric": "provider-prices-kimi-k3/Output",
          "label": "Kimi K3: price per million tokens by provider (Output)",
          "value": 15,
          "unit": "usd",
          "display": "$15.00",
          "context": "Kimi K3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-kimi-k3",
          "series": "Cache read",
          "point": "Fireworks",
          "metric": "provider-prices-kimi-k3/Cache read",
          "label": "Kimi K3: price per million tokens by provider (Cache read)",
          "value": 0.3,
          "unit": "usd",
          "display": "$0.30",
          "context": "Kimi K3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "series": "Input",
          "point": "Fireworks",
          "metric": "provider-prices-glm-5-3/Input",
          "label": "GLM 5.3: price per million tokens by provider (Input)",
          "value": 1.4,
          "unit": "usd",
          "display": "$1.40",
          "context": "GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "series": "Output",
          "point": "Fireworks",
          "metric": "provider-prices-glm-5-3/Output",
          "label": "GLM 5.3: price per million tokens by provider (Output)",
          "value": 4.4,
          "unit": "usd",
          "display": "$4.40",
          "context": "GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "series": "Cache read",
          "point": "Fireworks",
          "metric": "provider-prices-glm-5-3/Cache read",
          "label": "GLM 5.3: price per million tokens by provider (Cache read)",
          "value": 0.26,
          "unit": "usd",
          "display": "$0.26",
          "context": "GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "deepinfra",
      "name": "DeepInfra",
      "vendor": "DeepInfra",
      "kind": "provider",
      "description": "An inference provider listed on OpenRouter. Its prices here are the ones OpenRouter’s public API reported for its endpoints.",
      "aliases": [
        "DeepInfra"
      ],
      "facts": [
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "series": "Input",
          "point": "DeepInfra (bf16)",
          "metric": "provider-prices-gpt-oss-120b/Input",
          "label": "gpt-oss-120b: price per million tokens by provider (Input)",
          "value": 0.037,
          "unit": "usd",
          "display": "$0.037",
          "context": "bf16 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "series": "Output",
          "point": "DeepInfra (bf16)",
          "metric": "provider-prices-gpt-oss-120b/Output",
          "label": "gpt-oss-120b: price per million tokens by provider (Output)",
          "value": 0.17,
          "unit": "usd",
          "display": "$0.17",
          "context": "bf16 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-llama-3-3-70b-instruct",
          "series": "Input",
          "point": "DeepInfra (fp8)",
          "metric": "provider-prices-llama-3-3-70b-instruct/Input",
          "label": "Llama 3.3 70B Instruct: price per million tokens by provider (Input)",
          "value": 0.1,
          "unit": "usd",
          "display": "$0.10",
          "context": "fp8 · Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-llama-3-3-70b-instruct",
          "series": "Output",
          "point": "DeepInfra (fp8)",
          "metric": "provider-prices-llama-3-3-70b-instruct/Output",
          "label": "Llama 3.3 70B Instruct: price per million tokens by provider (Output)",
          "value": 0.32,
          "unit": "usd",
          "display": "$0.32",
          "context": "fp8 · Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-deepseek-v4-pro",
          "series": "Input",
          "point": "DeepInfra (fp8)",
          "metric": "provider-prices-deepseek-v4-pro/Input",
          "label": "DeepSeek V4 Pro 0423: price per million tokens by provider (Input)",
          "value": 1.3,
          "unit": "usd",
          "display": "$1.30",
          "context": "fp8 · DeepSeek V4 Pro 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-deepseek-v4-pro",
          "series": "Output",
          "point": "DeepInfra (fp8)",
          "metric": "provider-prices-deepseek-v4-pro/Output",
          "label": "DeepSeek V4 Pro 0423: price per million tokens by provider (Output)",
          "value": 2.6,
          "unit": "usd",
          "display": "$2.60",
          "context": "fp8 · DeepSeek V4 Pro 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-deepseek-v4-pro",
          "series": "Cache read",
          "point": "DeepInfra (fp8)",
          "metric": "provider-prices-deepseek-v4-pro/Cache read",
          "label": "DeepSeek V4 Pro 0423: price per million tokens by provider (Cache read)",
          "value": 0.1,
          "unit": "usd",
          "display": "$0.10",
          "context": "fp8 · DeepSeek V4 Pro 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-deepseek-v4-flash",
          "series": "Input",
          "point": "DeepInfra (fp8)",
          "metric": "provider-prices-deepseek-v4-flash/Input",
          "label": "DeepSeek V4 Flash 0423: price per million tokens by provider (Input)",
          "value": 0.09,
          "unit": "usd",
          "display": "$0.090",
          "context": "fp8 · DeepSeek V4 Flash 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-deepseek-v4-flash",
          "series": "Output",
          "point": "DeepInfra (fp8)",
          "metric": "provider-prices-deepseek-v4-flash/Output",
          "label": "DeepSeek V4 Flash 0423: price per million tokens by provider (Output)",
          "value": 0.18,
          "unit": "usd",
          "display": "$0.18",
          "context": "fp8 · DeepSeek V4 Flash 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-deepseek-v4-flash",
          "series": "Cache read",
          "point": "DeepInfra (fp8)",
          "metric": "provider-prices-deepseek-v4-flash/Cache read",
          "label": "DeepSeek V4 Flash 0423: price per million tokens by provider (Cache read)",
          "value": 0.018,
          "unit": "usd",
          "display": "$0.018",
          "context": "fp8 · DeepSeek V4 Flash 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-kimi-k3",
          "series": "Input",
          "point": "DeepInfra (mxfp4)",
          "metric": "provider-prices-kimi-k3/Input",
          "label": "Kimi K3: price per million tokens by provider (Input)",
          "value": 2.85,
          "unit": "usd",
          "display": "$2.85",
          "context": "mxfp4 · Kimi K3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-kimi-k3",
          "series": "Output",
          "point": "DeepInfra (mxfp4)",
          "metric": "provider-prices-kimi-k3/Output",
          "label": "Kimi K3: price per million tokens by provider (Output)",
          "value": 14.25,
          "unit": "usd",
          "display": "$14.25",
          "context": "mxfp4 · Kimi K3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-kimi-k3",
          "series": "Cache read",
          "point": "DeepInfra (mxfp4)",
          "metric": "provider-prices-kimi-k3/Cache read",
          "label": "Kimi K3: price per million tokens by provider (Cache read)",
          "value": 0.285,
          "unit": "usd",
          "display": "$0.28",
          "context": "mxfp4 · Kimi K3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "series": "Input",
          "point": "DeepInfra (fp4)",
          "metric": "provider-prices-glm-5-3/Input",
          "label": "GLM 5.3: price per million tokens by provider (Input)",
          "value": 0.5625,
          "unit": "usd",
          "display": "$0.56",
          "context": "fp4 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "series": "Output",
          "point": "DeepInfra (fp4)",
          "metric": "provider-prices-glm-5-3/Output",
          "label": "GLM 5.3: price per million tokens by provider (Output)",
          "value": 2.5,
          "unit": "usd",
          "display": "$2.50",
          "context": "fp4 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "series": "Cache read",
          "point": "DeepInfra (fp4)",
          "metric": "provider-prices-glm-5-3/Cache read",
          "label": "GLM 5.3: price per million tokens by provider (Cache read)",
          "value": 0.125,
          "unit": "usd",
          "display": "$0.13",
          "context": "fp4 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "cerebras",
      "name": "Cerebras",
      "vendor": "Cerebras",
      "kind": "provider",
      "description": "An inference provider listed on OpenRouter. Its prices here are the ones OpenRouter’s public API reported for its endpoints.",
      "aliases": [
        "Cerebras"
      ],
      "facts": [
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "series": "Input",
          "point": "Cerebras (fp16)",
          "metric": "provider-prices-gpt-oss-120b/Input",
          "label": "gpt-oss-120b: price per million tokens by provider (Input)",
          "value": 0.35,
          "unit": "usd",
          "display": "$0.35",
          "context": "fp16 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "series": "Output",
          "point": "Cerebras (fp16)",
          "metric": "provider-prices-gpt-oss-120b/Output",
          "label": "gpt-oss-120b: price per million tokens by provider (Output)",
          "value": 0.75,
          "unit": "usd",
          "display": "$0.75",
          "context": "fp16 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "series": "Cache read",
          "point": "Cerebras (fp16)",
          "metric": "provider-prices-gpt-oss-120b/Cache read",
          "label": "gpt-oss-120b: price per million tokens by provider (Cache read)",
          "value": 0.35,
          "unit": "usd",
          "display": "$0.35",
          "context": "fp16 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "sambanova",
      "name": "SambaNova",
      "vendor": "SambaNova",
      "kind": "provider",
      "description": "An inference provider listed on OpenRouter. Its prices here are the ones OpenRouter’s public API reported for its endpoints.",
      "aliases": [
        "SambaNova"
      ],
      "facts": [
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "series": "Input",
          "point": "SambaNova",
          "metric": "provider-prices-gpt-oss-120b/Input",
          "label": "gpt-oss-120b: price per million tokens by provider (Input)",
          "value": 0.14,
          "unit": "usd",
          "display": "$0.14",
          "context": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "series": "Output",
          "point": "SambaNova",
          "metric": "provider-prices-gpt-oss-120b/Output",
          "label": "gpt-oss-120b: price per million tokens by provider (Output)",
          "value": 0.95,
          "unit": "usd",
          "display": "$0.95",
          "context": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-llama-3-3-70b-instruct",
          "series": "Input",
          "point": "SambaNova",
          "metric": "provider-prices-llama-3-3-70b-instruct/Input",
          "label": "Llama 3.3 70B Instruct: price per million tokens by provider (Input)",
          "value": 0.45,
          "unit": "usd",
          "display": "$0.45",
          "context": "Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-llama-3-3-70b-instruct",
          "series": "Output",
          "point": "SambaNova",
          "metric": "provider-prices-llama-3-3-70b-instruct/Output",
          "label": "Llama 3.3 70B Instruct: price per million tokens by provider (Output)",
          "value": 0.9,
          "unit": "usd",
          "display": "$0.90",
          "context": "Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "nebius",
      "name": "Nebius",
      "vendor": "Nebius",
      "kind": "provider",
      "description": "An inference provider listed on OpenRouter. Its prices here are the ones OpenRouter’s public API reported for its endpoints.",
      "aliases": [
        "Nebius"
      ],
      "facts": [
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "series": "Input",
          "point": "Nebius (fp4)",
          "metric": "provider-prices-gpt-oss-120b/Input",
          "label": "gpt-oss-120b: price per million tokens by provider (Input)",
          "value": 0.15,
          "unit": "usd",
          "display": "$0.15",
          "context": "fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "series": "Output",
          "point": "Nebius (fp4)",
          "metric": "provider-prices-gpt-oss-120b/Output",
          "label": "gpt-oss-120b: price per million tokens by provider (Output)",
          "value": 0.6,
          "unit": "usd",
          "display": "$0.60",
          "context": "fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "series": "Input",
          "point": "Nebius (fp4)",
          "metric": "provider-prices-glm-5-3/Input",
          "label": "GLM 5.3: price per million tokens by provider (Input)",
          "value": 1.4,
          "unit": "usd",
          "display": "$1.40",
          "context": "fp4 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "series": "Output",
          "point": "Nebius (fp4)",
          "metric": "provider-prices-glm-5-3/Output",
          "label": "GLM 5.3: price per million tokens by provider (Output)",
          "value": 4.4,
          "unit": "usd",
          "display": "$4.40",
          "context": "fp4 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "parasail",
      "name": "Parasail",
      "vendor": "Parasail",
      "kind": "provider",
      "description": "An inference provider listed on OpenRouter. Its prices here are the ones OpenRouter’s public API reported for its endpoints.",
      "aliases": [
        "Parasail"
      ],
      "facts": [
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "series": "Input",
          "point": "Parasail (fp4)",
          "metric": "provider-prices-gpt-oss-120b/Input",
          "label": "gpt-oss-120b: price per million tokens by provider (Input)",
          "value": 0.1,
          "unit": "usd",
          "display": "$0.10",
          "context": "fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "series": "Output",
          "point": "Parasail (fp4)",
          "metric": "provider-prices-gpt-oss-120b/Output",
          "label": "gpt-oss-120b: price per million tokens by provider (Output)",
          "value": 0.75,
          "unit": "usd",
          "display": "$0.75",
          "context": "fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "series": "Cache read",
          "point": "Parasail (fp4)",
          "metric": "provider-prices-gpt-oss-120b/Cache read",
          "label": "gpt-oss-120b: price per million tokens by provider (Cache read)",
          "value": 0.055,
          "unit": "usd",
          "display": "$0.055",
          "context": "fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-llama-4-maverick",
          "series": "Input",
          "point": "Parasail (fp8)",
          "metric": "provider-prices-llama-4-maverick/Input",
          "label": "Llama 4 Maverick: price per million tokens by provider (Input)",
          "value": 0.35,
          "unit": "usd",
          "display": "$0.35",
          "context": "fp8 · Llama 4 Maverick · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-llama-4-maverick",
          "series": "Output",
          "point": "Parasail (fp8)",
          "metric": "provider-prices-llama-4-maverick/Output",
          "label": "Llama 4 Maverick: price per million tokens by provider (Output)",
          "value": 1,
          "unit": "usd",
          "display": "$1.00",
          "context": "fp8 · Llama 4 Maverick · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-llama-3-3-70b-instruct",
          "series": "Input",
          "point": "Parasail (fp8)",
          "metric": "provider-prices-llama-3-3-70b-instruct/Input",
          "label": "Llama 3.3 70B Instruct: price per million tokens by provider (Input)",
          "value": 0.22,
          "unit": "usd",
          "display": "$0.22",
          "context": "fp8 · Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-llama-3-3-70b-instruct",
          "series": "Output",
          "point": "Parasail (fp8)",
          "metric": "provider-prices-llama-3-3-70b-instruct/Output",
          "label": "Llama 3.3 70B Instruct: price per million tokens by provider (Output)",
          "value": 0.5,
          "unit": "usd",
          "display": "$0.50",
          "context": "fp8 · Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-llama-3-3-70b-instruct",
          "series": "Cache read",
          "point": "Parasail (fp8)",
          "metric": "provider-prices-llama-3-3-70b-instruct/Cache read",
          "label": "Llama 3.3 70B Instruct: price per million tokens by provider (Cache read)",
          "value": 0.11,
          "unit": "usd",
          "display": "$0.11",
          "context": "fp8 · Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-deepseek-v4-pro",
          "series": "Input",
          "point": "Parasail (fp8)",
          "metric": "provider-prices-deepseek-v4-pro/Input",
          "label": "DeepSeek V4 Pro 0423: price per million tokens by provider (Input)",
          "value": 0.45,
          "unit": "usd",
          "display": "$0.45",
          "context": "fp8 · DeepSeek V4 Pro 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-deepseek-v4-pro",
          "series": "Output",
          "point": "Parasail (fp8)",
          "metric": "provider-prices-deepseek-v4-pro/Output",
          "label": "DeepSeek V4 Pro 0423: price per million tokens by provider (Output)",
          "value": 3.48,
          "unit": "usd",
          "display": "$3.48",
          "context": "fp8 · DeepSeek V4 Pro 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-deepseek-v4-pro",
          "series": "Cache read",
          "point": "Parasail (fp8)",
          "metric": "provider-prices-deepseek-v4-pro/Cache read",
          "label": "DeepSeek V4 Pro 0423: price per million tokens by provider (Cache read)",
          "value": 0.1,
          "unit": "usd",
          "display": "$0.10",
          "context": "fp8 · DeepSeek V4 Pro 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-deepseek-v4-flash",
          "series": "Input",
          "point": "Parasail (fp8)",
          "metric": "provider-prices-deepseek-v4-flash/Input",
          "label": "DeepSeek V4 Flash 0423: price per million tokens by provider (Input)",
          "value": 0.14,
          "unit": "usd",
          "display": "$0.14",
          "context": "fp8 · DeepSeek V4 Flash 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-deepseek-v4-flash",
          "series": "Output",
          "point": "Parasail (fp8)",
          "metric": "provider-prices-deepseek-v4-flash/Output",
          "label": "DeepSeek V4 Flash 0423: price per million tokens by provider (Output)",
          "value": 0.28,
          "unit": "usd",
          "display": "$0.28",
          "context": "fp8 · DeepSeek V4 Flash 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-deepseek-v4-flash",
          "series": "Cache read",
          "point": "Parasail (fp8)",
          "metric": "provider-prices-deepseek-v4-flash/Cache read",
          "label": "DeepSeek V4 Flash 0423: price per million tokens by provider (Cache read)",
          "value": 0.07,
          "unit": "usd",
          "display": "$0.070",
          "context": "fp8 · DeepSeek V4 Flash 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-kimi-k3",
          "series": "Input",
          "point": "Parasail (fp4)",
          "metric": "provider-prices-kimi-k3/Input",
          "label": "Kimi K3: price per million tokens by provider (Input)",
          "value": 3,
          "unit": "usd",
          "display": "$3.00",
          "context": "fp4 · Kimi K3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-kimi-k3",
          "series": "Output",
          "point": "Parasail (fp4)",
          "metric": "provider-prices-kimi-k3/Output",
          "label": "Kimi K3: price per million tokens by provider (Output)",
          "value": 15,
          "unit": "usd",
          "display": "$15.00",
          "context": "fp4 · Kimi K3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-kimi-k3",
          "series": "Cache read",
          "point": "Parasail (fp4)",
          "metric": "provider-prices-kimi-k3/Cache read",
          "label": "Kimi K3: price per million tokens by provider (Cache read)",
          "value": 0.3,
          "unit": "usd",
          "display": "$0.30",
          "context": "fp4 · Kimi K3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "series": "Input",
          "point": "Parasail (fp8)",
          "metric": "provider-prices-glm-5-3/Input",
          "label": "GLM 5.3: price per million tokens by provider (Input)",
          "value": 1.4,
          "unit": "usd",
          "display": "$1.40",
          "context": "fp8 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "series": "Output",
          "point": "Parasail (fp8)",
          "metric": "provider-prices-glm-5-3/Output",
          "label": "GLM 5.3: price per million tokens by provider (Output)",
          "value": 4.4,
          "unit": "usd",
          "display": "$4.40",
          "context": "fp8 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "series": "Cache read",
          "point": "Parasail (fp8)",
          "metric": "provider-prices-glm-5-3/Cache read",
          "label": "GLM 5.3: price per million tokens by provider (Cache read)",
          "value": 0.26,
          "unit": "usd",
          "display": "$0.26",
          "context": "fp8 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "novita",
      "name": "Novita AI",
      "vendor": "Novita AI",
      "kind": "provider",
      "description": "An inference provider listed on OpenRouter. Its prices here are the ones OpenRouter’s public API reported for its endpoints.",
      "aliases": [
        "Novita"
      ],
      "facts": [
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "series": "Input",
          "point": "Novita (fp4)",
          "metric": "provider-prices-gpt-oss-120b/Input",
          "label": "gpt-oss-120b: price per million tokens by provider (Input)",
          "value": 0.05,
          "unit": "usd",
          "display": "$0.050",
          "context": "fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "series": "Output",
          "point": "Novita (fp4)",
          "metric": "provider-prices-gpt-oss-120b/Output",
          "label": "gpt-oss-120b: price per million tokens by provider (Output)",
          "value": 0.25,
          "unit": "usd",
          "display": "$0.25",
          "context": "fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-llama-4-maverick",
          "series": "Input",
          "point": "Novita (fp8)",
          "metric": "provider-prices-llama-4-maverick/Input",
          "label": "Llama 4 Maverick: price per million tokens by provider (Input)",
          "value": 0.27,
          "unit": "usd",
          "display": "$0.27",
          "context": "fp8 · Llama 4 Maverick · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-llama-4-maverick",
          "series": "Output",
          "point": "Novita (fp8)",
          "metric": "provider-prices-llama-4-maverick/Output",
          "label": "Llama 4 Maverick: price per million tokens by provider (Output)",
          "value": 0.85,
          "unit": "usd",
          "display": "$0.85",
          "context": "fp8 · Llama 4 Maverick · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-llama-3-3-70b-instruct",
          "series": "Input",
          "point": "Novita (bf16)",
          "metric": "provider-prices-llama-3-3-70b-instruct/Input",
          "label": "Llama 3.3 70B Instruct: price per million tokens by provider (Input)",
          "value": 0.135,
          "unit": "usd",
          "display": "$0.14",
          "context": "bf16 · Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-llama-3-3-70b-instruct",
          "series": "Output",
          "point": "Novita (bf16)",
          "metric": "provider-prices-llama-3-3-70b-instruct/Output",
          "label": "Llama 3.3 70B Instruct: price per million tokens by provider (Output)",
          "value": 0.4,
          "unit": "usd",
          "display": "$0.40",
          "context": "bf16 · Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-deepseek-v4-pro",
          "series": "Input",
          "point": "Novita (fp8)",
          "metric": "provider-prices-deepseek-v4-pro/Input",
          "label": "DeepSeek V4 Pro 0423: price per million tokens by provider (Input)",
          "value": 1.6,
          "unit": "usd",
          "display": "$1.60",
          "context": "fp8 · DeepSeek V4 Pro 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-deepseek-v4-pro",
          "series": "Output",
          "point": "Novita (fp8)",
          "metric": "provider-prices-deepseek-v4-pro/Output",
          "label": "DeepSeek V4 Pro 0423: price per million tokens by provider (Output)",
          "value": 3.2,
          "unit": "usd",
          "display": "$3.20",
          "context": "fp8 · DeepSeek V4 Pro 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-deepseek-v4-pro",
          "series": "Cache read",
          "point": "Novita (fp8)",
          "metric": "provider-prices-deepseek-v4-pro/Cache read",
          "label": "DeepSeek V4 Pro 0423: price per million tokens by provider (Cache read)",
          "value": 0.135,
          "unit": "usd",
          "display": "$0.14",
          "context": "fp8 · DeepSeek V4 Pro 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-deepseek-v4-flash",
          "series": "Input",
          "point": "Novita (fp8)",
          "metric": "provider-prices-deepseek-v4-flash/Input",
          "label": "DeepSeek V4 Flash 0423: price per million tokens by provider (Input)",
          "value": 0.14,
          "unit": "usd",
          "display": "$0.14",
          "context": "fp8 · DeepSeek V4 Flash 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-deepseek-v4-flash",
          "series": "Output",
          "point": "Novita (fp8)",
          "metric": "provider-prices-deepseek-v4-flash/Output",
          "label": "DeepSeek V4 Flash 0423: price per million tokens by provider (Output)",
          "value": 0.28,
          "unit": "usd",
          "display": "$0.28",
          "context": "fp8 · DeepSeek V4 Flash 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-deepseek-v4-flash",
          "series": "Cache read",
          "point": "Novita (fp8)",
          "metric": "provider-prices-deepseek-v4-flash/Cache read",
          "label": "DeepSeek V4 Flash 0423: price per million tokens by provider (Cache read)",
          "value": 0.028,
          "unit": "usd",
          "display": "$0.028",
          "context": "fp8 · DeepSeek V4 Flash 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "series": "Input",
          "point": "Novita (fp8)",
          "metric": "provider-prices-glm-5-3/Input",
          "label": "GLM 5.3: price per million tokens by provider (Input)",
          "value": 0.42,
          "unit": "usd",
          "display": "$0.42",
          "context": "fp8 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "series": "Output",
          "point": "Novita (fp8)",
          "metric": "provider-prices-glm-5-3/Output",
          "label": "GLM 5.3: price per million tokens by provider (Output)",
          "value": 1.32,
          "unit": "usd",
          "display": "$1.32",
          "context": "fp8 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "series": "Cache read",
          "point": "Novita (fp8)",
          "metric": "provider-prices-glm-5-3/Cache read",
          "label": "GLM 5.3: price per million tokens by provider (Cache read)",
          "value": 0.078,
          "unit": "usd",
          "display": "$0.078",
          "context": "fp8 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "baseten",
      "name": "Baseten",
      "vendor": "Baseten",
      "kind": "provider",
      "description": "An inference provider listed on OpenRouter. Its prices here are the ones OpenRouter’s public API reported for its endpoints.",
      "aliases": [
        "BaseTen"
      ],
      "facts": [
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "series": "Input",
          "point": "BaseTen (fp4)",
          "metric": "provider-prices-gpt-oss-120b/Input",
          "label": "gpt-oss-120b: price per million tokens by provider (Input)",
          "value": 0.1,
          "unit": "usd",
          "display": "$0.10",
          "context": "fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "series": "Output",
          "point": "BaseTen (fp4)",
          "metric": "provider-prices-gpt-oss-120b/Output",
          "label": "gpt-oss-120b: price per million tokens by provider (Output)",
          "value": 0.5,
          "unit": "usd",
          "display": "$0.50",
          "context": "fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "series": "Cache read",
          "point": "BaseTen (fp4)",
          "metric": "provider-prices-gpt-oss-120b/Cache read",
          "label": "gpt-oss-120b: price per million tokens by provider (Cache read)",
          "value": 0.1,
          "unit": "usd",
          "display": "$0.10",
          "context": "fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-kimi-k3",
          "series": "Input",
          "point": "BaseTen (fp8)",
          "metric": "provider-prices-kimi-k3/Input",
          "label": "Kimi K3: price per million tokens by provider (Input)",
          "value": 3,
          "unit": "usd",
          "display": "$3.00",
          "context": "fp8 · Kimi K3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-kimi-k3",
          "series": "Output",
          "point": "BaseTen (fp8)",
          "metric": "provider-prices-kimi-k3/Output",
          "label": "Kimi K3: price per million tokens by provider (Output)",
          "value": 15,
          "unit": "usd",
          "display": "$15.00",
          "context": "fp8 · Kimi K3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-kimi-k3",
          "series": "Cache read",
          "point": "BaseTen (fp8)",
          "metric": "provider-prices-kimi-k3/Cache read",
          "label": "Kimi K3: price per million tokens by provider (Cache read)",
          "value": 0.3,
          "unit": "usd",
          "display": "$0.30",
          "context": "fp8 · Kimi K3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "series": "Input",
          "point": "BaseTen (fp4)",
          "metric": "provider-prices-glm-5-3/Input",
          "label": "GLM 5.3: price per million tokens by provider (Input)",
          "value": 1.4,
          "unit": "usd",
          "display": "$1.40",
          "context": "fp4 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "series": "Output",
          "point": "BaseTen (fp4)",
          "metric": "provider-prices-glm-5-3/Output",
          "label": "GLM 5.3: price per million tokens by provider (Output)",
          "value": 4.4,
          "unit": "usd",
          "display": "$4.40",
          "context": "fp4 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "series": "Cache read",
          "point": "BaseTen (fp4)",
          "metric": "provider-prices-glm-5-3/Cache read",
          "label": "GLM 5.3: price per million tokens by provider (Cache read)",
          "value": 0.14,
          "unit": "usd",
          "display": "$0.14",
          "context": "fp4 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "cloudflare",
      "name": "Cloudflare Workers AI",
      "vendor": "Cloudflare",
      "kind": "provider",
      "description": "An inference provider listed on OpenRouter. Its prices here are the ones OpenRouter’s public API reported for its endpoints.",
      "aliases": [
        "Cloudflare"
      ],
      "facts": [
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-llama-3-3-70b-instruct",
          "series": "Input",
          "point": "Cloudflare (fp8)",
          "metric": "provider-prices-llama-3-3-70b-instruct/Input",
          "label": "Llama 3.3 70B Instruct: price per million tokens by provider (Input)",
          "value": 0.293,
          "unit": "usd",
          "display": "$0.29",
          "context": "fp8 · Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-llama-3-3-70b-instruct",
          "series": "Output",
          "point": "Cloudflare (fp8)",
          "metric": "provider-prices-llama-3-3-70b-instruct/Output",
          "label": "Llama 3.3 70B Instruct: price per million tokens by provider (Output)",
          "value": 2.253,
          "unit": "usd",
          "display": "$2.25",
          "context": "fp8 · Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-deepseek-v4-pro",
          "series": "Input",
          "point": "Cloudflare",
          "metric": "provider-prices-deepseek-v4-pro/Input",
          "label": "DeepSeek V4 Pro 0423: price per million tokens by provider (Input)",
          "value": 1.15,
          "unit": "usd",
          "display": "$1.15",
          "context": "DeepSeek V4 Pro 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-deepseek-v4-pro",
          "series": "Output",
          "point": "Cloudflare",
          "metric": "provider-prices-deepseek-v4-pro/Output",
          "label": "DeepSeek V4 Pro 0423: price per million tokens by provider (Output)",
          "value": 2.55,
          "unit": "usd",
          "display": "$2.55",
          "context": "DeepSeek V4 Pro 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-deepseek-v4-pro",
          "series": "Cache read",
          "point": "Cloudflare",
          "metric": "provider-prices-deepseek-v4-pro/Cache read",
          "label": "DeepSeek V4 Pro 0423: price per million tokens by provider (Cache read)",
          "value": 0.2,
          "unit": "usd",
          "display": "$0.20",
          "context": "DeepSeek V4 Pro 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-deepseek-v4-flash",
          "series": "Input",
          "point": "Cloudflare",
          "metric": "provider-prices-deepseek-v4-flash/Input",
          "label": "DeepSeek V4 Flash 0423: price per million tokens by provider (Input)",
          "value": 0.44,
          "unit": "usd",
          "display": "$0.44",
          "context": "DeepSeek V4 Flash 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-deepseek-v4-flash",
          "series": "Output",
          "point": "Cloudflare",
          "metric": "provider-prices-deepseek-v4-flash/Output",
          "label": "DeepSeek V4 Flash 0423: price per million tokens by provider (Output)",
          "value": 1.32,
          "unit": "usd",
          "display": "$1.32",
          "context": "DeepSeek V4 Flash 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-deepseek-v4-flash",
          "series": "Cache read",
          "point": "Cloudflare",
          "metric": "provider-prices-deepseek-v4-flash/Cache read",
          "label": "DeepSeek V4 Flash 0423: price per million tokens by provider (Cache read)",
          "value": 0.014,
          "unit": "usd",
          "display": "$0.014",
          "context": "DeepSeek V4 Flash 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "series": "Input",
          "point": "Cloudflare",
          "metric": "provider-prices-glm-5-3/Input",
          "label": "GLM 5.3: price per million tokens by provider (Input)",
          "value": 1.4,
          "unit": "usd",
          "display": "$1.40",
          "context": "GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "series": "Output",
          "point": "Cloudflare",
          "metric": "provider-prices-glm-5-3/Output",
          "label": "GLM 5.3: price per million tokens by provider (Output)",
          "value": 4.4,
          "unit": "usd",
          "display": "$4.40",
          "context": "GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "series": "Cache read",
          "point": "Cloudflare",
          "metric": "provider-prices-glm-5-3/Cache read",
          "label": "GLM 5.3: price per million tokens by provider (Cache read)",
          "value": 0.26,
          "unit": "usd",
          "display": "$0.26",
          "context": "GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "siliconflow",
      "name": "SiliconFlow",
      "vendor": "SiliconFlow",
      "kind": "provider",
      "description": "An inference provider listed on OpenRouter. Its prices here are the ones OpenRouter’s public API reported for its endpoints.",
      "aliases": [
        "SiliconFlow"
      ],
      "facts": [
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "series": "Input",
          "point": "SiliconFlow (fp8)",
          "metric": "provider-prices-gpt-oss-120b/Input",
          "label": "gpt-oss-120b: price per million tokens by provider (Input)",
          "value": 0.15,
          "unit": "usd",
          "display": "$0.15",
          "context": "fp8 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "series": "Output",
          "point": "SiliconFlow (fp8)",
          "metric": "provider-prices-gpt-oss-120b/Output",
          "label": "gpt-oss-120b: price per million tokens by provider (Output)",
          "value": 0.6,
          "unit": "usd",
          "display": "$0.60",
          "context": "fp8 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "series": "Cache read",
          "point": "SiliconFlow (fp8)",
          "metric": "provider-prices-gpt-oss-120b/Cache read",
          "label": "gpt-oss-120b: price per million tokens by provider (Cache read)",
          "value": 0.075,
          "unit": "usd",
          "display": "$0.075",
          "context": "fp8 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-deepseek-v4-pro",
          "series": "Input",
          "point": "SiliconFlow (fp8)",
          "metric": "provider-prices-deepseek-v4-pro/Input",
          "label": "DeepSeek V4 Pro 0423: price per million tokens by provider (Input)",
          "value": 1.50162,
          "unit": "usd",
          "display": "$1.50",
          "context": "fp8 · DeepSeek V4 Pro 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-deepseek-v4-pro",
          "series": "Output",
          "point": "SiliconFlow (fp8)",
          "metric": "provider-prices-deepseek-v4-pro/Output",
          "label": "DeepSeek V4 Pro 0423: price per million tokens by provider (Output)",
          "value": 3.135,
          "unit": "usd",
          "display": "$3.13",
          "context": "fp8 · DeepSeek V4 Pro 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-deepseek-v4-pro",
          "series": "Cache read",
          "point": "SiliconFlow (fp8)",
          "metric": "provider-prices-deepseek-v4-pro/Cache read",
          "label": "DeepSeek V4 Pro 0423: price per million tokens by provider (Cache read)",
          "value": 0.135,
          "unit": "usd",
          "display": "$0.14",
          "context": "fp8 · DeepSeek V4 Pro 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-deepseek-v4-flash",
          "series": "Input",
          "point": "SiliconFlow (fp8)",
          "metric": "provider-prices-deepseek-v4-flash/Input",
          "label": "DeepSeek V4 Flash 0423: price per million tokens by provider (Input)",
          "value": 0.13,
          "unit": "usd",
          "display": "$0.13",
          "context": "fp8 · DeepSeek V4 Flash 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-deepseek-v4-flash",
          "series": "Output",
          "point": "SiliconFlow (fp8)",
          "metric": "provider-prices-deepseek-v4-flash/Output",
          "label": "DeepSeek V4 Flash 0423: price per million tokens by provider (Output)",
          "value": 0.28,
          "unit": "usd",
          "display": "$0.28",
          "context": "fp8 · DeepSeek V4 Flash 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-deepseek-v4-flash",
          "series": "Cache read",
          "point": "SiliconFlow (fp8)",
          "metric": "provider-prices-deepseek-v4-flash/Cache read",
          "label": "DeepSeek V4 Flash 0423: price per million tokens by provider (Cache read)",
          "value": 0.028,
          "unit": "usd",
          "display": "$0.028",
          "context": "fp8 · DeepSeek V4 Flash 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "series": "Input",
          "point": "SiliconFlow (fp8)",
          "metric": "provider-prices-glm-5-3/Input",
          "label": "GLM 5.3: price per million tokens by provider (Input)",
          "value": 0.7,
          "unit": "usd",
          "display": "$0.70",
          "context": "fp8 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "series": "Output",
          "point": "SiliconFlow (fp8)",
          "metric": "provider-prices-glm-5-3/Output",
          "label": "GLM 5.3: price per million tokens by provider (Output)",
          "value": 2.2,
          "unit": "usd",
          "display": "$2.20",
          "context": "fp8 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "series": "Cache read",
          "point": "SiliconFlow (fp8)",
          "metric": "provider-prices-glm-5-3/Cache read",
          "label": "GLM 5.3: price per million tokens by provider (Cache read)",
          "value": 0.13,
          "unit": "usd",
          "display": "$0.13",
          "context": "fp8 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    }
  ],
  "comparisons": [
    {
      "slug": "claude-sonnet-5-5-vs-claude-opus-5-5",
      "a": "claude-sonnet-5-5",
      "b": "claude-opus-5-5",
      "title": "Claude Sonnet 5.5 vs Claude Opus 5.5",
      "seoTitle": "Claude Sonnet 5.5 vs Claude Opus 5.5: measured benchmarks",
      "description": "Claude Sonnet 5.5 vs Claude Opus 5.5: 35 measured metrics from 8 studies, with sample sizes, intervals and every failure counted.",
      "verdict": "Claude Sonnet 5.5 and Claude Opus 5.5 share 35 measured metrics and 31 list-price calculations from 10 studies. No row separates them: every interval or run range overlaps, too few runs were recorded, no interval was recorded, or more is not better for that metric. The rows are 16 ties and 50 unclear; each row says why. Calculation rows are derived from list prices and recorded counts; they are not bills or runs. Some rows rest on small samples (n = 3 at the smallest).",
      "rows": [
        {
          "metric": "Resolved on the same 3 SWE-bench Verified instances (interim)",
          "aValue": 0.3333,
          "bValue": 0.6667,
          "unit": "rate",
          "aDisplay": "33% (1/3)",
          "bDisplay": "67% (2/3)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Sonnet 5.5 6% to 79%; Claude Opus 5.5 21% to 94%), so this sample cannot separate them.",
          "studySlug": "swe-bench-opus-vs-sonnet",
          "n": 3,
          "chartId": "swebench-opus-sonnet-resolved",
          "aN": 3,
          "bN": 3,
          "aContext": "Agent · older builds · SWE-bench Verified, interim paired probe",
          "bContext": "Agent · new build · SWE-bench Verified, interim paired probe",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.0615,
            0.7923
          ],
          "bRange": [
            0.2077,
            0.9385
          ]
        },
        {
          "metric": "List-price cost per attempt (calculation)",
          "aValue": 2.88,
          "bValue": 7.59,
          "unit": "usd",
          "aDisplay": "$2.88",
          "bDisplay": "$7.59",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($2.88 vs $7.59, 2.6x) is not tested against run-to-run variation.",
          "studySlug": "swe-bench-opus-vs-sonnet",
          "n": 3,
          "chartId": "swebench-opus-sonnet-cost-per-attempt",
          "aN": 3,
          "bN": 3,
          "aContext": "Agent · older builds · SWE-bench Verified, interim paired probe",
          "bContext": "Agent · new build · SWE-bench Verified, interim paired probe",
          "calculation": true
        },
        {
          "metric": "Worker time per attempt",
          "aValue": 9.37,
          "bValue": 20.26,
          "unit": "minutes",
          "aDisplay": "9.4 min",
          "bDisplay": "20.3 min",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Sonnet 5.5 4.7 min to 15.0 min; Claude Opus 5.5 9.8 min to 25.3 min); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "swe-bench-opus-vs-sonnet",
          "n": 3,
          "chartId": "swebench-opus-sonnet-minutes",
          "aN": 3,
          "bN": 3,
          "aContext": "Agent · older builds · SWE-bench Verified, interim paired probe",
          "bContext": "Agent · new build · SWE-bench Verified, interim paired probe",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            4.74,
            15
          ],
          "bRange": [
            9.76,
            25.29
          ]
        },
        {
          "metric": "Pass rate on five validated tasks",
          "aValue": 0.8,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "80% (12/15)",
          "bDisplay": "100% (15/15)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Sonnet 5.5 55% to 93%; Claude Opus 5.5 80% to 100%), so this sample cannot separate them.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-pass-rate",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code · five short validated tasks",
          "bContext": "Claude Code · five short validated tasks",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.5481,
            0.9295
          ],
          "bRange": [
            0.7961,
            1
          ]
        },
        {
          "metric": "Total time per call",
          "aValue": 2.31,
          "bValue": 2.75,
          "unit": "seconds",
          "aDisplay": "2.31 s",
          "bDisplay": "2.75 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Sonnet 5.5 2.17 s to 7.73 s; Claude Opus 5.5 2.47 s to 8.91 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-total-latency",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code · five short validated tasks",
          "bContext": "Claude Code · five short validated tasks",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            2.17,
            7.73
          ],
          "bRange": [
            2.47,
            8.91
          ]
        },
        {
          "metric": "Time to first useful output",
          "aValue": 1.56,
          "bValue": 1.92,
          "unit": "seconds",
          "aDisplay": "1.56 s",
          "bDisplay": "1.92 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Sonnet 5.5 0.99 s to 6.39 s; Claude Opus 5.5 1.56 s to 7.23 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-first-useful-latency",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code · five short validated tasks",
          "bContext": "Claude Code · five short validated tasks",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            0.99,
            6.39
          ],
          "bRange": [
            1.56,
            7.23
          ]
        },
        {
          "metric": "Input tokens per call: what the CLI sends (Cache read)",
          "aValue": 1401,
          "bValue": 1401,
          "unit": "tokens",
          "aDisplay": "1,401",
          "bDisplay": "1,401",
          "winner": "tie",
          "basis": "Same value. More or fewer is not better by itself for this metric.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-input-tokens",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code · five short validated tasks",
          "bContext": "Claude Code · five short validated tasks"
        },
        {
          "metric": "Input tokens per call: what the CLI sends (Other input)",
          "aValue": 685,
          "bValue": 680,
          "unit": "tokens",
          "aDisplay": "685",
          "bDisplay": "680",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-input-tokens",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code · five short validated tasks",
          "bContext": "Claude Code · five short validated tasks"
        },
        {
          "metric": "Output tokens per call (Output tokens)",
          "aValue": 107,
          "bValue": 64,
          "unit": "tokens",
          "aDisplay": "107",
          "bDisplay": "64",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-output-tokens",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code · five short validated tasks",
          "bContext": "Claude Code · five short validated tasks"
        },
        {
          "metric": "List-price cost per call (calculation)",
          "aValue": 0.0036,
          "bValue": 0.00688,
          "unit": "usd",
          "aDisplay": "$0.0036",
          "bDisplay": "$0.0069",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Sonnet 5.5 $0.0034 to $0.010; Claude Opus 5.5 $0.0059 to $0.022); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-list-price-per-call",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code · five short validated tasks",
          "bContext": "Claude Code · five short validated tasks",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            0.00342,
            0.01021
          ],
          "bRange": [
            0.00592,
            0.02226
          ],
          "calculation": true
        },
        {
          "metric": "List-price cost per passing answer (calculation)",
          "aValue": 0.00624,
          "bValue": 0.01009,
          "unit": "usd",
          "aDisplay": "$0.0062",
          "bDisplay": "$0.010",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.0062 vs $0.010) is not tested against run-to-run variation.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-cost-per-pass",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code · five short validated tasks",
          "bContext": "Claude Code · five short validated tasks",
          "calculation": true
        },
        {
          "metric": "Pass rate on eight hard tasks (Strict pass)",
          "aValue": 1,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "100% (24/24)",
          "bDisplay": "100% (24/24)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Sonnet 5.5 86% to 100%; Claude Opus 5.5 86% to 100%), so this sample cannot separate them.",
          "studySlug": "hard-model-head-to-head",
          "n": 24,
          "chartId": "hard-h2h-pass-rate",
          "aN": 24,
          "bN": 24,
          "aContext": "Claude Code · eight hard validated tasks",
          "bContext": "Claude Code · eight hard validated tasks",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.862,
            1
          ],
          "bRange": [
            0.862,
            1
          ]
        },
        {
          "metric": "Pass rate on eight hard tasks (Lenient (format misses counted))",
          "aValue": 1,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "100% (24/24)",
          "bDisplay": "100% (24/24)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Sonnet 5.5 86% to 100%; Claude Opus 5.5 86% to 100%), so this sample cannot separate them.",
          "studySlug": "hard-model-head-to-head",
          "n": 24,
          "chartId": "hard-h2h-pass-rate",
          "aN": 24,
          "bN": 24,
          "aContext": "Claude Code · eight hard validated tasks",
          "bContext": "Claude Code · eight hard validated tasks",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.862,
            1
          ],
          "bRange": [
            0.862,
            1
          ]
        },
        {
          "metric": "Total time per call on hard tasks (separate batches)",
          "aValue": 7.75,
          "bValue": 9.18,
          "unit": "seconds",
          "aDisplay": "7.75 s",
          "bDisplay": "9.18 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Sonnet 5.5 2.26 s to 34.8 s; Claude Opus 5.5 4.24 s to 27.2 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "hard-model-head-to-head",
          "n": 24,
          "chartId": "hard-h2h-total-latency",
          "aN": 24,
          "bN": 24,
          "aContext": "Claude Code · eight hard validated tasks",
          "bContext": "Claude Code · eight hard validated tasks",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            2.26,
            34.79
          ],
          "bRange": [
            4.24,
            27.21
          ]
        },
        {
          "metric": "Time to first useful output on hard tasks",
          "aValue": 5.95,
          "bValue": 6.78,
          "unit": "seconds",
          "aDisplay": "5.95 s",
          "bDisplay": "6.78 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Sonnet 5.5 0.86 s to 30.6 s; Claude Opus 5.5 2.39 s to 21.8 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "hard-model-head-to-head",
          "n": 24,
          "chartId": "hard-h2h-first-useful-latency",
          "aN": 24,
          "bN": 24,
          "aContext": "Claude Code · eight hard validated tasks",
          "bContext": "Claude Code · eight hard validated tasks",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            0.86,
            30.57
          ],
          "bRange": [
            2.39,
            21.77
          ]
        },
        {
          "metric": "Output tokens per call on hard tasks (Output tokens)",
          "aValue": 1050,
          "bValue": 945,
          "unit": "tokens",
          "aDisplay": "1,050",
          "bDisplay": "945",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "hard-model-head-to-head",
          "n": 24,
          "chartId": "hard-h2h-output-tokens",
          "aN": 24,
          "bN": 24,
          "aContext": "Claude Code · eight hard validated tasks",
          "bContext": "Claude Code · eight hard validated tasks"
        },
        {
          "metric": "List-price cost per strict pass on hard tasks (calculation)",
          "aValue": 0.01435,
          "bValue": 0.02824,
          "unit": "usd",
          "aDisplay": "$0.014",
          "bDisplay": "$0.028",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.014 vs $0.028) is not tested against run-to-run variation.",
          "studySlug": "hard-model-head-to-head",
          "n": 24,
          "chartId": "hard-h2h-cost-per-pass",
          "aN": 24,
          "bN": 24,
          "aContext": "Claude Code · eight hard validated tasks",
          "bContext": "Claude Code · eight hard validated tasks",
          "calculation": true
        },
        {
          "metric": "Coding sessions that passed every hidden check",
          "aValue": 1,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "100% (12/12)",
          "bDisplay": "100% (12/12)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Sonnet 5.5 76% to 100%; Claude Opus 5.5 76% to 100%), so this sample cannot separate them.",
          "studySlug": "coding-agents-head-to-head",
          "n": 12,
          "chartId": "coding-agents-pass-rate",
          "aN": 12,
          "bN": 12,
          "aContext": "Claude Code · six small repository tasks with hidden tests",
          "bContext": "Claude Code · six small repository tasks with hidden tests",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.7575,
            1
          ],
          "bRange": [
            0.7575,
            1
          ]
        },
        {
          "metric": "Time per coding session",
          "aValue": 23.1,
          "bValue": 56.9,
          "unit": "seconds",
          "aDisplay": "23.1 s",
          "bDisplay": "56.9 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Sonnet 5.5 18.7 s to 44.5 s; Claude Opus 5.5 29.8 s to 185.8 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "coding-agents-head-to-head",
          "n": 12,
          "chartId": "coding-agents-wall-time",
          "aN": 12,
          "bN": 12,
          "aContext": "Claude Code · six small repository tasks with hidden tests",
          "bContext": "Claude Code · six small repository tasks with hidden tests",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            18.7,
            44.5
          ],
          "bRange": [
            29.8,
            185.8
          ]
        },
        {
          "metric": "Tool calls per coding session",
          "aValue": 7.5,
          "bValue": 7.5,
          "unit": "calls",
          "aDisplay": "7.5",
          "bDisplay": "7.5",
          "winner": "tie",
          "basis": "Same value. More or fewer is not better by itself for this metric.",
          "studySlug": "coding-agents-head-to-head",
          "n": 12,
          "chartId": "coding-agents-tool-calls",
          "aN": 12,
          "bN": 12,
          "aContext": "Claude Code · six small repository tasks with hidden tests",
          "bContext": "Claude Code · six small repository tasks with hidden tests",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            3,
            14
          ],
          "bRange": [
            5,
            14
          ]
        },
        {
          "metric": "List-price cost per passing coding session (calculation)",
          "aValue": 0.085,
          "bValue": 0.2229,
          "unit": "usd",
          "aDisplay": "$0.085",
          "bDisplay": "$0.22",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.085 vs $0.22, 2.6x) is not tested against run-to-run variation.",
          "studySlug": "coding-agents-head-to-head",
          "n": 12,
          "chartId": "coding-agents-cost-per-pass",
          "aN": 12,
          "bN": 12,
          "aContext": "Claude Code · six small repository tasks with hidden tests",
          "bContext": "Claude Code · six small repository tasks with hidden tests",
          "calculation": true
        },
        {
          "metric": "Strict pass rate by effort on eight hard tasks",
          "aValue": 1,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "100% (16/16)",
          "bDisplay": "100% (16/16)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Sonnet 5.5 81% to 100%; Claude Opus 5.5 81% to 100%), so this sample cannot separate them.",
          "studySlug": "effort-ladder",
          "n": 16,
          "chartId": "effort-ladder-pass-rate",
          "aN": 16,
          "bN": 16,
          "aContext": "Claude Code · eight hard validated tasks, effort ladder",
          "bContext": "Claude Code · eight hard validated tasks, effort ladder",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.8064,
            1
          ],
          "bRange": [
            0.8064,
            1
          ]
        },
        {
          "metric": "Total time per call by effort on hard tasks",
          "aValue": 7.97,
          "bValue": 9.18,
          "unit": "seconds",
          "aDisplay": "7.97 s",
          "bDisplay": "9.18 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Sonnet 5.5 2.26 s to 21.6 s; Claude Opus 5.5 4.24 s to 27.2 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "effort-ladder",
          "n": 16,
          "chartId": "effort-ladder-total-latency",
          "aN": 16,
          "bN": 16,
          "aContext": "Claude Code · eight hard validated tasks, effort ladder",
          "bContext": "Claude Code · eight hard validated tasks, effort ladder",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            2.26,
            21.61
          ],
          "bRange": [
            4.24,
            27.21
          ]
        },
        {
          "metric": "Output tokens per call by effort on hard tasks (Output tokens)",
          "aValue": 1054,
          "bValue": 945,
          "unit": "tokens",
          "aDisplay": "1,054",
          "bDisplay": "945",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "effort-ladder",
          "n": 16,
          "chartId": "effort-ladder-output-tokens",
          "aN": 16,
          "bN": 16,
          "aContext": "Claude Code · eight hard validated tasks, effort ladder",
          "bContext": "Claude Code · eight hard validated tasks, effort ladder"
        },
        {
          "metric": "List-price cost per strict pass by effort (calculation)",
          "aValue": 0.01398,
          "bValue": 0.02893,
          "unit": "usd",
          "aDisplay": "$0.014",
          "bDisplay": "$0.029",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.014 vs $0.029, 2.1x) is not tested against run-to-run variation.",
          "studySlug": "effort-ladder",
          "n": 16,
          "chartId": "effort-ladder-cost-per-pass",
          "aN": 16,
          "bN": 16,
          "aContext": "Claude Code · eight hard validated tasks, effort ladder",
          "bContext": "Claude Code · eight hard validated tasks, effort ladder",
          "calculation": true
        },
        {
          "metric": "List-price cost of 5-question sessions with and without the cache (calculation) (With the cache, as recorded)",
          "aValue": 0.135003,
          "bValue": 0.255057,
          "unit": "usd",
          "aDisplay": "$0.14",
          "bDisplay": "$0.26",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.14 vs $0.26) is not tested against run-to-run variation.",
          "studySlug": "caching-consistency",
          "n": 15,
          "chartId": "caching-cost-with-without",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code · calculation: 5-turn cached sessions over a fixed ledger",
          "bContext": "Claude Code · calculation: 5-turn cached sessions over a fixed ledger",
          "calculation": true
        },
        {
          "metric": "List-price cost of 5-question sessions with and without the cache (calculation) (Without a cache: every input token at the input price)",
          "aValue": 0.269788,
          "bValue": 0.544228,
          "unit": "usd",
          "aDisplay": "$0.27",
          "bDisplay": "$0.54",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.27 vs $0.54, 2.0x) is not tested against run-to-run variation.",
          "studySlug": "caching-consistency",
          "n": 15,
          "chartId": "caching-cost-with-without",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code · calculation: 5-turn cached sessions over a fixed ledger",
          "bContext": "Claude Code · calculation: 5-turn cached sessions over a fixed ledger",
          "calculation": true
        },
        {
          "metric": "Time per turn: first turn vs later turns in a cached session (Turn 1 (writes the ledger to the cache))",
          "aValue": 1.64,
          "bValue": 1.9,
          "unit": "seconds",
          "aDisplay": "1.64 s",
          "bDisplay": "1.90 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Sonnet 5.5 1.58 s to 1.79 s; Claude Opus 5.5 1.78 s to 4.36 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "caching-consistency",
          "n": 3,
          "chartId": "caching-latency-first-vs-later",
          "aN": 3,
          "bN": 3,
          "aContext": "Claude Code · 5-turn cached sessions over a fixed ledger",
          "bContext": "Claude Code · 5-turn cached sessions over a fixed ledger",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            1.58,
            1.79
          ],
          "bRange": [
            1.78,
            4.36
          ]
        },
        {
          "metric": "Time per turn: first turn vs later turns in a cached session (Turns 2-5 (read the ledger from the cache))",
          "aValue": 1.61,
          "bValue": 2.4,
          "unit": "seconds",
          "aDisplay": "1.61 s",
          "bDisplay": "2.40 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Sonnet 5.5 1.35 s to 5.63 s; Claude Opus 5.5 1.63 s to 12.7 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "caching-consistency",
          "n": 12,
          "chartId": "caching-latency-first-vs-later",
          "aN": 12,
          "bN": 12,
          "aContext": "Claude Code · 5-turn cached sessions over a fixed ledger",
          "bContext": "Claude Code · 5-turn cached sessions over a fixed ledger",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            1.35,
            5.63
          ],
          "bRange": [
            1.63,
            12.67
          ]
        },
        {
          "metric": "Cost of a reused prefix with and without the cache, by session length (calculation): 1 turn",
          "aValue": 15.66,
          "bValue": 31.31,
          "unit": "usd",
          "aDisplay": "$15.66",
          "bDisplay": "$31.31",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($15.66 vs $31.31) is not tested against run-to-run variation.",
          "studySlug": "prompt-cache-break-even",
          "chartId": "cache-break-even-cost-curve",
          "aContext": "no cache",
          "bContext": "no cache",
          "calculation": true
        },
        {
          "metric": "Cost of a reused prefix with and without the cache, by session length (calculation): 2 turns",
          "aValue": 31.32,
          "bValue": 62.62,
          "unit": "usd",
          "aDisplay": "$31.32",
          "bDisplay": "$62.62",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($31.32 vs $62.62) is not tested against run-to-run variation.",
          "studySlug": "prompt-cache-break-even",
          "chartId": "cache-break-even-cost-curve",
          "aContext": "no cache",
          "bContext": "no cache",
          "calculation": true
        },
        {
          "metric": "Cost of a reused prefix with and without the cache, by session length (calculation): 3 turns",
          "aValue": 46.99,
          "bValue": 93.94,
          "unit": "usd",
          "aDisplay": "$46.99",
          "bDisplay": "$93.94",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($46.99 vs $93.94) is not tested against run-to-run variation.",
          "studySlug": "prompt-cache-break-even",
          "chartId": "cache-break-even-cost-curve",
          "aContext": "no cache",
          "bContext": "no cache",
          "calculation": true
        },
        {
          "metric": "Cost of a reused prefix with and without the cache, by session length (calculation): 5 turns",
          "aValue": 78.31,
          "bValue": 156.56,
          "unit": "usd",
          "aDisplay": "$78.31",
          "bDisplay": "$156.56",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($78.31 vs $156.56) is not tested against run-to-run variation.",
          "studySlug": "prompt-cache-break-even",
          "chartId": "cache-break-even-cost-curve",
          "aContext": "no cache",
          "bContext": "no cache",
          "calculation": true
        },
        {
          "metric": "Cost of a reused prefix with and without the cache, by session length (calculation): 10 turns",
          "aValue": 156.62,
          "bValue": 313.12,
          "unit": "usd",
          "aDisplay": "$156.62",
          "bDisplay": "$313.12",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($156.62 vs $313.12) is not tested against run-to-run variation.",
          "studySlug": "prompt-cache-break-even",
          "chartId": "cache-break-even-cost-curve",
          "aContext": "no cache",
          "bContext": "no cache",
          "calculation": true
        },
        {
          "metric": "Cost of a reused prefix with and without the cache, by session length (calculation): 20 turns",
          "aValue": 313.24,
          "bValue": 626.24,
          "unit": "usd",
          "aDisplay": "$313.24",
          "bDisplay": "$626.24",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($313.24 vs $626.24) is not tested against run-to-run variation.",
          "studySlug": "prompt-cache-break-even",
          "chartId": "cache-break-even-cost-curve",
          "aContext": "no cache",
          "bContext": "no cache",
          "calculation": true
        },
        {
          "metric": "One 10-turn session or ten 1-turn sessions: cost with and without the cache (calculation) (No cache (the same either way))",
          "aValue": 156.62,
          "bValue": 313.12,
          "unit": "usd",
          "aDisplay": "$156.62",
          "bDisplay": "$313.12",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($156.62 vs $313.12) is not tested against run-to-run variation.",
          "studySlug": "prompt-cache-break-even",
          "chartId": "cache-break-even-session-split",
          "aContext": "",
          "bContext": "cache read $0.2 per M",
          "calculation": true
        },
        {
          "metric": "One 10-turn session or ten 1-turn sessions: cost with and without the cache (calculation) (One 10-turn session, 1-hour cache)",
          "aValue": 45.42,
          "bValue": 76.71,
          "unit": "usd",
          "aDisplay": "$45.42",
          "bDisplay": "$76.71",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($45.42 vs $76.71) is not tested against run-to-run variation.",
          "studySlug": "prompt-cache-break-even",
          "chartId": "cache-break-even-session-split",
          "aContext": "",
          "bContext": "cache read $0.2 per M",
          "calculation": true
        },
        {
          "metric": "One 10-turn session or ten 1-turn sessions: cost with and without the cache (calculation) (Ten 1-turn sessions, 1-hour cache (assumed no reuse of the new prefix))",
          "aValue": 313.24,
          "bValue": 626.24,
          "unit": "usd",
          "aDisplay": "$313.24",
          "bDisplay": "$626.24",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($313.24 vs $626.24) is not tested against run-to-run variation.",
          "studySlug": "prompt-cache-break-even",
          "chartId": "cache-break-even-session-split",
          "aContext": "",
          "bContext": "cache read $0.2 per M",
          "calculation": true
        },
        {
          "metric": "Reasoning share of output tokens per call on hard tasks (calculation)",
          "aValue": 54.54,
          "bValue": 54.79,
          "unit": "percent",
          "aDisplay": "54.5%",
          "bDisplay": "54.8%",
          "winner": "unclear",
          "basis": "More or fewer percent is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "n": 24,
          "chartId": "thinking-bill-share",
          "aN": 24,
          "bN": 24,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            0,
            95.91
          ],
          "bRange": [
            29.92,
            95.6
          ],
          "calculation": true
        },
        {
          "metric": "List-price cost per call: reasoning, remaining output and input (calculation) (Reasoning (output tokens))",
          "aValue": 0.006665,
          "bValue": 0.012528,
          "unit": "usd",
          "aDisplay": "$0.0067",
          "bDisplay": "$0.013",
          "winner": "unclear",
          "basis": "More or fewer usd is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "n": 24,
          "chartId": "thinking-bill-cost-per-call",
          "aN": 24,
          "bN": 24,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "calculation": true
        },
        {
          "metric": "List-price cost per call: reasoning, remaining output and input (calculation) (Remaining output (visible-answer estimate))",
          "aValue": 0.003672,
          "bValue": 0.008003,
          "unit": "usd",
          "aDisplay": "$0.0037",
          "bDisplay": "$0.0080",
          "winner": "unclear",
          "basis": "More or fewer usd is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "n": 24,
          "chartId": "thinking-bill-cost-per-call",
          "aN": 24,
          "bN": 24,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "calculation": true
        },
        {
          "metric": "List-price cost per call: reasoning, remaining output and input (calculation) (Input (prompt, cache priced))",
          "aValue": 0.004012,
          "bValue": 0.00771,
          "unit": "usd",
          "aDisplay": "$0.0040",
          "bDisplay": "$0.0077",
          "winner": "unclear",
          "basis": "More or fewer usd is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "n": 24,
          "chartId": "thinking-bill-cost-per-call",
          "aN": 24,
          "bN": 24,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "calculation": true
        },
        {
          "metric": "Reasoning cost per strict pass by effort, with the total (calculation) (Reasoning cost per strict pass)",
          "aValue": 0.006299,
          "bValue": 0.013104,
          "unit": "usd",
          "aDisplay": "$0.0063",
          "bDisplay": "$0.013",
          "winner": "unclear",
          "basis": "More or fewer usd is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "n": 16,
          "chartId": "thinking-bill-by-effort",
          "aN": 16,
          "bN": 16,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "calculation": true
        },
        {
          "metric": "Reasoning cost per strict pass by effort, with the total (calculation) (Total cost per strict pass)",
          "aValue": 0.013978,
          "bValue": 0.028925,
          "unit": "usd",
          "aDisplay": "$0.014",
          "bDisplay": "$0.029",
          "winner": "unclear",
          "basis": "More or fewer usd is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "n": 16,
          "chartId": "thinking-bill-by-effort",
          "aN": 16,
          "bN": 16,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "calculation": true
        },
        {
          "metric": "Reasoning share on short tasks vs hard tasks (calculation) (Eight hard tasks)",
          "aValue": 54.54,
          "bValue": 54.79,
          "unit": "percent",
          "aDisplay": "54.5%",
          "bDisplay": "54.8%",
          "winner": "unclear",
          "basis": "More or fewer percent is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "n": 24,
          "chartId": "thinking-bill-short-vs-hard",
          "aN": 24,
          "bN": 24,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            0,
            95.91
          ],
          "bRange": [
            29.92,
            95.6
          ],
          "calculation": true
        },
        {
          "metric": "Reasoning share on short tasks vs hard tasks (calculation) (Five short tasks)",
          "aValue": 0,
          "bValue": 0,
          "unit": "percent",
          "aDisplay": "0%",
          "bDisplay": "0%",
          "winner": "tie",
          "basis": "Same value. More or fewer is not better by itself for this metric.",
          "studySlug": "thinking-token-bill",
          "n": 15,
          "chartId": "thinking-bill-short-vs-hard",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            0,
            72.75
          ],
          "bRange": [
            0,
            93.33
          ],
          "calculation": true
        },
        {
          "metric": "Time to first text: a 250-line answer, six models",
          "aValue": 1.96,
          "bValue": 1.97,
          "unit": "seconds",
          "aDisplay": "1.96 s",
          "bDisplay": "1.97 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Sonnet 5.5 0.88 s to 4.09 s; Claude Opus 5.5 1.70 s to 2.35 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "llm-speed-anatomy",
          "n": 4,
          "chartId": "speed-anatomy-first-text",
          "aN": 4,
          "bN": 4,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            0.88,
            4.09
          ],
          "bRange": [
            1.7,
            2.35
          ]
        },
        {
          "metric": "Output speed after the first text: visible tokens per second (calculation)",
          "aValue": 231.7,
          "bValue": 155.5,
          "unit": "tokens",
          "aDisplay": "232",
          "bDisplay": "156",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "llm-speed-anatomy",
          "n": 4,
          "chartId": "speed-anatomy-output-speed",
          "aN": 4,
          "bN": 4,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            230.3,
            233
          ],
          "bRange": [
            154.6,
            156.4
          ],
          "calculation": true
        },
        {
          "metric": "Output speed in characters per second after the first text (calculation)",
          "aValue": 517,
          "bValue": 347,
          "unit": "count",
          "aDisplay": "517",
          "bDisplay": "347",
          "winner": "unclear",
          "basis": "More or fewer count is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "llm-speed-anatomy",
          "n": 4,
          "chartId": "speed-anatomy-chars-per-second",
          "aN": 4,
          "bN": 4,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            513,
            519
          ],
          "bRange": [
            345,
            349
          ],
          "calculation": true
        },
        {
          "metric": "Time to first text as the prompt grows: 1k",
          "aValue": 1.45,
          "bValue": 1.51,
          "unit": "seconds",
          "aDisplay": "1.45 s",
          "bDisplay": "1.51 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Sonnet 5.5 1.23 s to 1.72 s; Claude Opus 5.5 1.46 s to 2.01 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "llm-speed-anatomy",
          "n": 3,
          "chartId": "speed-anatomy-prompt-size",
          "aN": 3,
          "bN": 3,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            1.23,
            1.72
          ],
          "bRange": [
            1.46,
            2.01
          ],
          "calculation": true
        },
        {
          "metric": "Time to first text as the prompt grows: 16k",
          "aValue": 1.78,
          "bValue": 1.74,
          "unit": "seconds",
          "aDisplay": "1.78 s",
          "bDisplay": "1.74 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Sonnet 5.5 1.64 s to 2.11 s; Claude Opus 5.5 1.70 s to 2.97 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "llm-speed-anatomy",
          "n": 3,
          "chartId": "speed-anatomy-prompt-size",
          "aN": 3,
          "bN": 3,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            1.64,
            2.11
          ],
          "bRange": [
            1.7,
            2.97
          ],
          "calculation": true
        },
        {
          "metric": "Time to first text as the prompt grows: 64k",
          "aValue": 3.07,
          "bValue": 1.79,
          "unit": "seconds",
          "aDisplay": "3.07 s",
          "bDisplay": "1.79 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Sonnet 5.5 1.38 s to 3.61 s; Claude Opus 5.5 1.72 s to 3.72 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "llm-speed-anatomy",
          "n": 3,
          "chartId": "speed-anatomy-prompt-size",
          "aN": 3,
          "bN": 3,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            1.38,
            3.61
          ],
          "bRange": [
            1.72,
            3.72
          ],
          "calculation": true
        },
        {
          "metric": "Total time per call by prompt size (1k prompt)",
          "aValue": 1.78,
          "bValue": 1.83,
          "unit": "seconds",
          "aDisplay": "1.78 s",
          "bDisplay": "1.83 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Sonnet 5.5 1.57 s to 2.12 s; Claude Opus 5.5 1.82 s to 2.41 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "llm-speed-anatomy",
          "n": 3,
          "chartId": "speed-anatomy-total-by-size",
          "aN": 3,
          "bN": 3,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            1.57,
            2.12
          ],
          "bRange": [
            1.82,
            2.41
          ]
        },
        {
          "metric": "Total time per call by prompt size (16k prompt)",
          "aValue": 2.1,
          "bValue": 2.36,
          "unit": "seconds",
          "aDisplay": "2.10 s",
          "bDisplay": "2.36 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Sonnet 5.5 1.98 s to 2.48 s; Claude Opus 5.5 2.11 s to 3.40 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "llm-speed-anatomy",
          "n": 3,
          "chartId": "speed-anatomy-total-by-size",
          "aN": 3,
          "bN": 3,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            1.98,
            2.48
          ],
          "bRange": [
            2.11,
            3.4
          ]
        },
        {
          "metric": "Total time per call by prompt size (64k prompt)",
          "aValue": 3.44,
          "bValue": 2.35,
          "unit": "seconds",
          "aDisplay": "3.44 s",
          "bDisplay": "2.35 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Sonnet 5.5 1.74 s to 4.38 s; Claude Opus 5.5 2.26 s to 4.29 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "llm-speed-anatomy",
          "n": 3,
          "chartId": "speed-anatomy-total-by-size",
          "aN": 3,
          "bN": 3,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            1.74,
            4.38
          ],
          "bRange": [
            2.26,
            4.29
          ]
        },
        {
          "metric": "Exact lookup answers at the 1k, 16k and 64k prompt-size targets",
          "aValue": 1,
          "bValue": 0.5556,
          "unit": "rate",
          "aDisplay": "100% (9/9)",
          "bDisplay": "56% (5/9)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Sonnet 5.5 70% to 100%; Claude Opus 5.5 27% to 81%), so this sample cannot separate them.",
          "studySlug": "llm-speed-anatomy",
          "n": 9,
          "chartId": "speed-anatomy-lookup-correct",
          "aN": 9,
          "bN": 9,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.7009,
            1
          ],
          "bRange": [
            0.2667,
            0.8112
          ]
        },
        {
          "metric": "Pass rate on 4 harder tasks (Strict pass)",
          "aValue": 0.375,
          "bValue": 0.4167,
          "unit": "rate",
          "aDisplay": "38% (6/16)",
          "bDisplay": "42% (5/12)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Sonnet 5.5 18% to 61%; Claude Opus 5.5 19% to 68%), so this sample cannot separate them.",
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-pass-rate",
          "aN": 16,
          "bN": 12,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.1848,
            0.6136
          ],
          "bRange": [
            0.1933,
            0.6805
          ]
        },
        {
          "metric": "Pass rate on 4 harder tasks (Lenient (format misses counted))",
          "aValue": 0.375,
          "bValue": 0.5,
          "unit": "rate",
          "aDisplay": "38% (6/16)",
          "bDisplay": "50% (6/12)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Sonnet 5.5 18% to 61%; Claude Opus 5.5 25% to 75%), so this sample cannot separate them.",
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-pass-rate",
          "aN": 16,
          "bN": 12,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.1848,
            0.6136
          ],
          "bRange": [
            0.2538,
            0.7462
          ]
        },
        {
          "metric": "Calls that tried a tool although tools were off",
          "aValue": 0.3125,
          "bValue": 0.4167,
          "unit": "rate",
          "aDisplay": "31% (5/16)",
          "bDisplay": "42% (5/12)",
          "winner": "unclear",
          "basis": "More or fewer rate is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-tool-attempts",
          "aN": 16,
          "bN": 12,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.1416,
            0.556
          ],
          "bRange": [
            0.1933,
            0.6805
          ]
        },
        {
          "metric": "Strict pass rate by task: 10x10 nonogram",
          "aValue": 1,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "100% (4/4)",
          "bDisplay": "100% (3/3)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Sonnet 5.5 51% to 100%; Claude Opus 5.5 44% to 100%), so this sample cannot separate them.",
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-pass-by-task",
          "aN": 4,
          "bN": 3,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.5101,
            1
          ],
          "bRange": [
            0.4385,
            1
          ]
        },
        {
          "metric": "Strict pass rate by task: Sudoku, 22 givens",
          "aValue": 0,
          "bValue": 0,
          "unit": "rate",
          "aDisplay": "0% (0/4)",
          "bDisplay": "0% (0/3)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Sonnet 5.5 0% to 49%; Claude Opus 5.5 0% to 56%), so this sample cannot separate them.",
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-pass-by-task",
          "aN": 4,
          "bN": 3,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0,
            0.4899
          ],
          "bRange": [
            0,
            0.5615
          ]
        },
        {
          "metric": "Strict pass rate by task: 6x6 Skyscrapers",
          "aValue": 0,
          "bValue": 0.3333,
          "unit": "rate",
          "aDisplay": "0% (0/4)",
          "bDisplay": "33% (1/3)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Sonnet 5.5 0% to 49%; Claude Opus 5.5 6% to 79%), so this sample cannot separate them.",
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-pass-by-task",
          "aN": 4,
          "bN": 3,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0,
            0.4899
          ],
          "bRange": [
            0.0615,
            0.7923
          ]
        },
        {
          "metric": "Strict pass rate by task: Seeded shuffle output",
          "aValue": 0.5,
          "bValue": 0.3333,
          "unit": "rate",
          "aDisplay": "50% (2/4)",
          "bDisplay": "33% (1/3)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Sonnet 5.5 15% to 85%; Claude Opus 5.5 6% to 79%), so this sample cannot separate them.",
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-pass-by-task",
          "aN": 4,
          "bN": 3,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.15,
            0.85
          ],
          "bRange": [
            0.0615,
            0.7923
          ]
        },
        {
          "metric": "Total time per call on harder tasks",
          "aValue": 70.43,
          "bValue": 80.34,
          "unit": "seconds",
          "aDisplay": "70.4 s",
          "bDisplay": "80.3 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Sonnet 5.5 4.32 s to 210.1 s; Claude Opus 5.5 3.82 s to 279.5 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-total-latency",
          "aN": 12,
          "bN": 9,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            4.32,
            210.08
          ],
          "bRange": [
            3.82,
            279.5
          ]
        },
        {
          "metric": "Output tokens per call on harder tasks (Output tokens)",
          "aValue": 9287,
          "bValue": 8420,
          "unit": "tokens",
          "aDisplay": "9,287",
          "bDisplay": "8,420",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-output-tokens",
          "aN": 12,
          "bN": 9,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            407,
            27921
          ],
          "bRange": [
            279,
            40044
          ]
        },
        {
          "metric": "List-price cost per strict pass on harder tasks (calculation)",
          "aValue": 0.23843,
          "bValue": 0.59333,
          "unit": "usd",
          "aDisplay": "$0.24",
          "bDisplay": "$0.59",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.24 vs $0.59, 2.5x) is not tested against run-to-run variation.",
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-cost-per-pass",
          "aN": 16,
          "bN": 12,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "calculation": true
        }
      ]
    },
    {
      "slug": "claude-haiku-4-5-vs-claude-sonnet-5-5",
      "a": "claude-haiku-4-5",
      "b": "claude-sonnet-5-5",
      "title": "Claude Haiku 4.5 vs Claude Sonnet 5.5",
      "seoTitle": "Claude Haiku 4.5 vs Claude Sonnet 5.5: measured benchmarks",
      "description": "Claude Haiku 4.5 vs Claude Sonnet 5.5: 113 measured metrics from 12 studies (Pass rate on five validated tasks; more), with sample sizes and intervals.",
      "verdict": "Claude Haiku 4.5 and Claude Sonnet 5.5 share 113 measured metrics and 40 list-price calculations from 14 studies. Claude Sonnet 5.5 leads on 21 rows: Pass rate on eight hard tasks (Strict pass), 100% (24/24) vs 46% (11/24); Pass rate on eight hard tasks (Lenient (format misses counted)), 100% (24/24) vs 67% (16/24); Same prompt, 10 times: strict pass rate (Exact number), 100% (10/10) vs 0% (0/10); and 18 more. On those rows the 95% intervals, run ranges and p50–p95 bands do not overlap; only a 95% interval is a confidence interval. The other rows are 58 ties and 74 unclear; each row says why. Calculation rows are derived from list prices and recorded counts; they are not bills or runs. Some rows rest on small samples (n = 2 at the smallest).",
      "rows": [
        {
          "metric": "Pass rate on five validated tasks",
          "aValue": 1,
          "bValue": 0.8,
          "unit": "rate",
          "aDisplay": "100% (15/15)",
          "bDisplay": "80% (12/15)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Haiku 4.5 80% to 100%; Claude Sonnet 5.5 55% to 93%), so this sample cannot separate them.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-pass-rate",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code · five short validated tasks",
          "bContext": "Claude Code · five short validated tasks",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.7961,
            1
          ],
          "bRange": [
            0.5481,
            0.9295
          ]
        },
        {
          "metric": "Total time per call",
          "aValue": 4.43,
          "bValue": 2.31,
          "unit": "seconds",
          "aDisplay": "4.43 s",
          "bDisplay": "2.31 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Haiku 4.5 3.16 s to 23.6 s; Claude Sonnet 5.5 2.17 s to 7.73 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-total-latency",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code · five short validated tasks",
          "bContext": "Claude Code · five short validated tasks",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            3.16,
            23.57
          ],
          "bRange": [
            2.17,
            7.73
          ]
        },
        {
          "metric": "Time to first useful output",
          "aValue": 3.63,
          "bValue": 1.56,
          "unit": "seconds",
          "aDisplay": "3.63 s",
          "bDisplay": "1.56 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Haiku 4.5 2.78 s to 22.3 s; Claude Sonnet 5.5 0.99 s to 6.39 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-first-useful-latency",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code · five short validated tasks",
          "bContext": "Claude Code · five short validated tasks",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            2.78,
            22.27
          ],
          "bRange": [
            0.99,
            6.39
          ]
        },
        {
          "metric": "Input tokens per call: what the CLI sends (Cache read)",
          "aValue": 0,
          "bValue": 1401,
          "unit": "tokens",
          "aDisplay": "0",
          "bDisplay": "1,401",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-input-tokens",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code · five short validated tasks",
          "bContext": "Claude Code · five short validated tasks"
        },
        {
          "metric": "Input tokens per call: what the CLI sends (Other input)",
          "aValue": 3790,
          "bValue": 685,
          "unit": "tokens",
          "aDisplay": "3,790",
          "bDisplay": "685",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-input-tokens",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code · five short validated tasks",
          "bContext": "Claude Code · five short validated tasks"
        },
        {
          "metric": "Output tokens per call (Output tokens)",
          "aValue": 367,
          "bValue": 107,
          "unit": "tokens",
          "aDisplay": "367",
          "bDisplay": "107",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-output-tokens",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code · five short validated tasks",
          "bContext": "Claude Code · five short validated tasks"
        },
        {
          "metric": "List-price cost per call (calculation)",
          "aValue": 0.00566,
          "bValue": 0.0036,
          "unit": "usd",
          "aDisplay": "$0.0057",
          "bDisplay": "$0.0036",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Haiku 4.5 $0.0051 to $0.018; Claude Sonnet 5.5 $0.0034 to $0.010); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-list-price-per-call",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code · five short validated tasks",
          "bContext": "Claude Code · five short validated tasks",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            0.00513,
            0.01804
          ],
          "bRange": [
            0.00342,
            0.01021
          ],
          "calculation": true
        },
        {
          "metric": "List-price cost per passing answer (calculation)",
          "aValue": 0.00836,
          "bValue": 0.00624,
          "unit": "usd",
          "aDisplay": "$0.0084",
          "bDisplay": "$0.0062",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.0084 vs $0.0062) is not tested against run-to-run variation.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-cost-per-pass",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code · five short validated tasks",
          "bContext": "Claude Code · five short validated tasks",
          "calculation": true
        },
        {
          "metric": "Pass rate on eight hard tasks (Strict pass)",
          "aValue": 0.4583,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "46% (11/24)",
          "bDisplay": "100% (24/24)",
          "winner": "b",
          "basis": "The 95% intervals do not overlap (Claude Haiku 4.5 28% to 65%; Claude Sonnet 5.5 86% to 100%).",
          "studySlug": "hard-model-head-to-head",
          "n": 24,
          "chartId": "hard-h2h-pass-rate",
          "aN": 24,
          "bN": 24,
          "aContext": "Claude Code · eight hard validated tasks",
          "bContext": "Claude Code · eight hard validated tasks",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.2789,
            0.6493
          ],
          "bRange": [
            0.862,
            1
          ]
        },
        {
          "metric": "Pass rate on eight hard tasks (Lenient (format misses counted))",
          "aValue": 0.6667,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "67% (16/24)",
          "bDisplay": "100% (24/24)",
          "winner": "b",
          "basis": "The 95% intervals do not overlap (Claude Haiku 4.5 47% to 82%; Claude Sonnet 5.5 86% to 100%).",
          "studySlug": "hard-model-head-to-head",
          "n": 24,
          "chartId": "hard-h2h-pass-rate",
          "aN": 24,
          "bN": 24,
          "aContext": "Claude Code · eight hard validated tasks",
          "bContext": "Claude Code · eight hard validated tasks",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.4671,
            0.8203
          ],
          "bRange": [
            0.862,
            1
          ]
        },
        {
          "metric": "Total time per call on hard tasks (separate batches)",
          "aValue": 39.01,
          "bValue": 7.75,
          "unit": "seconds",
          "aDisplay": "39.0 s",
          "bDisplay": "7.75 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Haiku 4.5 15.3 s to 75.1 s; Claude Sonnet 5.5 2.26 s to 34.8 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "hard-model-head-to-head",
          "n": 24,
          "chartId": "hard-h2h-total-latency",
          "aN": 24,
          "bN": 24,
          "aContext": "Claude Code · eight hard validated tasks",
          "bContext": "Claude Code · eight hard validated tasks",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            15.27,
            75.13
          ],
          "bRange": [
            2.26,
            34.79
          ]
        },
        {
          "metric": "Time to first useful output on hard tasks",
          "aValue": 35.54,
          "bValue": 5.95,
          "unit": "seconds",
          "aDisplay": "35.5 s",
          "bDisplay": "5.95 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Haiku 4.5 12.9 s to 70.3 s; Claude Sonnet 5.5 0.86 s to 30.6 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "hard-model-head-to-head",
          "n": 24,
          "chartId": "hard-h2h-first-useful-latency",
          "aN": 24,
          "bN": 24,
          "aContext": "Claude Code · eight hard validated tasks",
          "bContext": "Claude Code · eight hard validated tasks",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            12.88,
            70.31
          ],
          "bRange": [
            0.86,
            30.57
          ]
        },
        {
          "metric": "Output tokens per call on hard tasks (Output tokens)",
          "aValue": 5064,
          "bValue": 1050,
          "unit": "tokens",
          "aDisplay": "5,064",
          "bDisplay": "1,050",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "hard-model-head-to-head",
          "n": 24,
          "chartId": "hard-h2h-output-tokens",
          "aN": 24,
          "bN": 24,
          "aContext": "Claude Code · eight hard validated tasks",
          "bContext": "Claude Code · eight hard validated tasks"
        },
        {
          "metric": "List-price cost per strict pass on hard tasks (calculation)",
          "aValue": 0.0672,
          "bValue": 0.01435,
          "unit": "usd",
          "aDisplay": "$0.067",
          "bDisplay": "$0.014",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.067 vs $0.014, 4.7x) is not tested against run-to-run variation.",
          "studySlug": "hard-model-head-to-head",
          "n": 24,
          "chartId": "hard-h2h-cost-per-pass",
          "aN": 24,
          "bN": 24,
          "aContext": "Claude Code · eight hard validated tasks",
          "bContext": "Claude Code · eight hard validated tasks",
          "calculation": true
        },
        {
          "metric": "Same prompt, 10 times: strict pass rate (Exact number)",
          "aValue": 0,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "0% (0/10)",
          "bDisplay": "100% (10/10)",
          "winner": "b",
          "basis": "The 95% intervals do not overlap (Claude Haiku 4.5 0% to 28%; Claude Sonnet 5.5 72% to 100%).",
          "studySlug": "caching-consistency",
          "n": 10,
          "chartId": "consistency-pass-rate",
          "aN": 10,
          "bN": 10,
          "aContext": "Claude Code · same prompt repeated 10 times",
          "bContext": "Claude Code · same prompt repeated 10 times",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0,
            0.2775
          ],
          "bRange": [
            0.7225,
            1
          ]
        },
        {
          "metric": "Same prompt, 10 times: strict pass rate (JSON object)",
          "aValue": 0.1,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "10% (1/10)",
          "bDisplay": "100% (10/10)",
          "winner": "b",
          "basis": "The 95% intervals do not overlap (Claude Haiku 4.5 2% to 40%; Claude Sonnet 5.5 72% to 100%).",
          "studySlug": "caching-consistency",
          "n": 10,
          "chartId": "consistency-pass-rate",
          "aN": 10,
          "bN": 10,
          "aContext": "Claude Code · same prompt repeated 10 times",
          "bContext": "Claude Code · same prompt repeated 10 times",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.0179,
            0.4042
          ],
          "bRange": [
            0.7225,
            1
          ]
        },
        {
          "metric": "Same prompt, 10 times: strict pass rate (Code fix)",
          "aValue": 1,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "100% (10/10)",
          "bDisplay": "100% (10/10)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Haiku 4.5 72% to 100%; Claude Sonnet 5.5 72% to 100%), so this sample cannot separate them.",
          "studySlug": "caching-consistency",
          "n": 10,
          "chartId": "consistency-pass-rate",
          "aN": 10,
          "bN": 10,
          "aContext": "Claude Code · same prompt repeated 10 times",
          "bContext": "Claude Code · same prompt repeated 10 times",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.7225,
            1
          ],
          "bRange": [
            0.7225,
            1
          ]
        },
        {
          "metric": "Same prompt, 10 times: how many different answers (Exact number)",
          "aValue": 1,
          "bValue": 1,
          "unit": "count",
          "aDisplay": "1",
          "bDisplay": "1",
          "winner": "tie",
          "basis": "Same value. More or fewer is not better by itself for this metric.",
          "studySlug": "caching-consistency",
          "n": 10,
          "chartId": "consistency-distinct-answers",
          "aN": 10,
          "bN": 10,
          "aContext": "Claude Code · same prompt repeated 10 times",
          "bContext": "Claude Code · same prompt repeated 10 times"
        },
        {
          "metric": "Same prompt, 10 times: how many different answers (JSON object)",
          "aValue": 1,
          "bValue": 1,
          "unit": "count",
          "aDisplay": "1",
          "bDisplay": "1",
          "winner": "tie",
          "basis": "Same value. More or fewer is not better by itself for this metric.",
          "studySlug": "caching-consistency",
          "n": 10,
          "chartId": "consistency-distinct-answers",
          "aN": 10,
          "bN": 10,
          "aContext": "Claude Code · same prompt repeated 10 times",
          "bContext": "Claude Code · same prompt repeated 10 times"
        },
        {
          "metric": "Same prompt, 10 times: how many different answers (Code fix)",
          "aValue": 6,
          "bValue": 3,
          "unit": "count",
          "aDisplay": "6",
          "bDisplay": "3",
          "winner": "unclear",
          "basis": "More or fewer count is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "caching-consistency",
          "n": 10,
          "chartId": "consistency-distinct-answers",
          "aN": 10,
          "bN": 10,
          "aContext": "Claude Code · same prompt repeated 10 times",
          "bContext": "Claude Code · same prompt repeated 10 times"
        },
        {
          "metric": "Same prompt, 10 times: time per call (Exact number)",
          "aValue": 5.06,
          "bValue": 6.89,
          "unit": "seconds",
          "aDisplay": "5.06 s",
          "bDisplay": "6.89 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Haiku 4.5 4.42 s to 6.20 s; Claude Sonnet 5.5 5.81 s to 7.81 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "caching-consistency",
          "n": 10,
          "chartId": "consistency-latency-spread",
          "aN": 10,
          "bN": 10,
          "aContext": "Claude Code · same prompt repeated 10 times",
          "bContext": "Claude Code · same prompt repeated 10 times",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            4.42,
            6.2
          ],
          "bRange": [
            5.81,
            7.81
          ]
        },
        {
          "metric": "Same prompt, 10 times: time per call (JSON object)",
          "aValue": 7.03,
          "bValue": 2.89,
          "unit": "seconds",
          "aDisplay": "7.03 s",
          "bDisplay": "2.89 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Haiku 4.5 5.28 s to 12.3 s; Claude Sonnet 5.5 2.68 s to 5.30 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "caching-consistency",
          "n": 10,
          "chartId": "consistency-latency-spread",
          "aN": 10,
          "bN": 10,
          "aContext": "Claude Code · same prompt repeated 10 times",
          "bContext": "Claude Code · same prompt repeated 10 times",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            5.28,
            12.27
          ],
          "bRange": [
            2.68,
            5.3
          ]
        },
        {
          "metric": "Same prompt, 10 times: time per call (Code fix)",
          "aValue": 5.95,
          "bValue": 2.67,
          "unit": "seconds",
          "aDisplay": "5.95 s",
          "bDisplay": "2.67 s",
          "winner": "b",
          "basis": "The run ranges (fastest to slowest) do not overlap (Claude Haiku 4.5 4.89 s to 7.33 s; Claude Sonnet 5.5 2.32 s to 4.34 s). A range is not a confidence interval.",
          "studySlug": "caching-consistency",
          "n": 10,
          "chartId": "consistency-latency-spread",
          "aN": 10,
          "bN": 10,
          "aContext": "Claude Code · same prompt repeated 10 times",
          "bContext": "Claude Code · same prompt repeated 10 times",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            4.89,
            7.33
          ],
          "bRange": [
            2.32,
            4.34
          ]
        },
        {
          "metric": "Full pass rate by kind of memory: No memory",
          "aValue": 0.2,
          "bValue": 0.6,
          "unit": "rate",
          "aDisplay": "20% (2/10)",
          "bDisplay": "60% (9/15)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Haiku 4.5 6% to 51%; Claude Sonnet 5.5 36% to 80%), so this sample cannot separate them.",
          "studySlug": "agent-memory",
          "chartId": "memory-full-pass",
          "aN": 10,
          "bN": 15,
          "aContext": "",
          "bContext": "",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.0567,
            0.5098
          ],
          "bRange": [
            0.3575,
            0.8018
          ]
        },
        {
          "metric": "Full pass rate by kind of memory: /init CLAUDE.md",
          "aValue": 0.2,
          "bValue": 0.6,
          "unit": "rate",
          "aDisplay": "20% (2/10)",
          "bDisplay": "60% (9/15)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Haiku 4.5 6% to 51%; Claude Sonnet 5.5 36% to 80%), so this sample cannot separate them.",
          "studySlug": "agent-memory",
          "chartId": "memory-full-pass",
          "aN": 10,
          "bN": 15,
          "aContext": "",
          "bContext": "",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.0567,
            0.5098
          ],
          "bRange": [
            0.3575,
            0.8018
          ]
        },
        {
          "metric": "Full pass rate by kind of memory: Curated, 11 lines",
          "aValue": 0.7,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "70% (7/10)",
          "bDisplay": "100% (15/15)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Haiku 4.5 40% to 89%; Claude Sonnet 5.5 80% to 100%), so this sample cannot separate them.",
          "studySlug": "agent-memory",
          "chartId": "memory-full-pass",
          "aN": 10,
          "bN": 15,
          "aContext": "",
          "bContext": "",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.3968,
            0.8922
          ],
          "bRange": [
            0.7961,
            1
          ]
        },
        {
          "metric": "Full pass rate by kind of memory: Raw notes, 60 lines",
          "aValue": 0.6,
          "bValue": 0.9333,
          "unit": "rate",
          "aDisplay": "60% (6/10)",
          "bDisplay": "93% (14/15)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Haiku 4.5 31% to 83%; Claude Sonnet 5.5 70% to 99%), so this sample cannot separate them.",
          "studySlug": "agent-memory",
          "chartId": "memory-full-pass",
          "aN": 10,
          "bN": 15,
          "aContext": "",
          "bContext": "",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.3127,
            0.8318
          ],
          "bRange": [
            0.7018,
            0.9881
          ]
        },
        {
          "metric": "Full pass rate by kind of memory: Dreamed notes",
          "aValue": 0.7,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "70% (7/10)",
          "bDisplay": "100% (15/15)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Haiku 4.5 40% to 89%; Claude Sonnet 5.5 80% to 100%), so this sample cannot separate them.",
          "studySlug": "agent-memory",
          "chartId": "memory-full-pass",
          "aN": 10,
          "bN": 15,
          "aContext": "",
          "bContext": "",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.3968,
            0.8922
          ],
          "bRange": [
            0.7961,
            1
          ]
        },
        {
          "metric": "Full pass rate by kind of memory: Handbook, 210 lines",
          "aValue": 0.3,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "30% (3/10)",
          "bDisplay": "100% (15/15)",
          "winner": "b",
          "basis": "The 95% intervals do not overlap (Claude Haiku 4.5 11% to 60%; Claude Sonnet 5.5 80% to 100%).",
          "studySlug": "agent-memory",
          "chartId": "memory-full-pass",
          "aN": 10,
          "bN": 15,
          "aContext": "",
          "bContext": "",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.1078,
            0.6032
          ],
          "bRange": [
            0.7961,
            1
          ]
        },
        {
          "metric": "Full pass rate by kind of memory: Stop hook only",
          "aValue": 0.8,
          "bValue": 0.8,
          "unit": "rate",
          "aDisplay": "80% (8/10)",
          "bDisplay": "80% (12/15)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Haiku 4.5 49% to 94%; Claude Sonnet 5.5 55% to 93%), so this sample cannot separate them.",
          "studySlug": "agent-memory",
          "chartId": "memory-full-pass",
          "aN": 10,
          "bN": 15,
          "aContext": "",
          "bContext": "",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.4902,
            0.9433
          ],
          "bRange": [
            0.5481,
            0.9295
          ]
        },
        {
          "metric": "Full pass rate by kind of memory: Curated + hook",
          "aValue": 0.9,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "90% (9/10)",
          "bDisplay": "100% (15/15)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Haiku 4.5 60% to 98%; Claude Sonnet 5.5 80% to 100%), so this sample cannot separate them.",
          "studySlug": "agent-memory",
          "chartId": "memory-full-pass",
          "aN": 10,
          "bN": 15,
          "aContext": "",
          "bContext": "",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.5958,
            0.9821
          ],
          "bRange": [
            0.7961,
            1
          ]
        },
        {
          "metric": "Team knowledge followed, Sonnet vs Haiku: No memory",
          "aValue": 0,
          "bValue": 0.4,
          "unit": "rate",
          "aDisplay": "0% (0/10)",
          "bDisplay": "40% (6/15)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Haiku 4.5 0% to 28%; Claude Sonnet 5.5 20% to 64%), so this sample cannot separate them.",
          "studySlug": "agent-memory",
          "chartId": "memory-team-knowledge-by-model",
          "aN": 10,
          "bN": 15,
          "aContext": "",
          "bContext": "",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0,
            0.2775
          ],
          "bRange": [
            0.1982,
            0.6425
          ]
        },
        {
          "metric": "Team knowledge followed, Sonnet vs Haiku: /init CLAUDE.md",
          "aValue": 0.1,
          "bValue": 0.6667,
          "unit": "rate",
          "aDisplay": "10% (1/10)",
          "bDisplay": "67% (10/15)",
          "winner": "b",
          "basis": "The 95% intervals do not overlap (Claude Haiku 4.5 2% to 40%; Claude Sonnet 5.5 42% to 85%).",
          "studySlug": "agent-memory",
          "chartId": "memory-team-knowledge-by-model",
          "aN": 10,
          "bN": 15,
          "aContext": "",
          "bContext": "",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.0179,
            0.4042
          ],
          "bRange": [
            0.4171,
            0.8482
          ]
        },
        {
          "metric": "Team knowledge followed, Sonnet vs Haiku: Curated, 11 lines",
          "aValue": 0.8,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "80% (8/10)",
          "bDisplay": "100% (15/15)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Haiku 4.5 49% to 94%; Claude Sonnet 5.5 80% to 100%), so this sample cannot separate them.",
          "studySlug": "agent-memory",
          "chartId": "memory-team-knowledge-by-model",
          "aN": 10,
          "bN": 15,
          "aContext": "",
          "bContext": "",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.4902,
            0.9433
          ],
          "bRange": [
            0.7961,
            1
          ]
        },
        {
          "metric": "Team knowledge followed, Sonnet vs Haiku: Raw notes, 60 lines",
          "aValue": 0.6,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "60% (6/10)",
          "bDisplay": "100% (15/15)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Haiku 4.5 31% to 83%; Claude Sonnet 5.5 80% to 100%), so this sample cannot separate them.",
          "studySlug": "agent-memory",
          "chartId": "memory-team-knowledge-by-model",
          "aN": 10,
          "bN": 15,
          "aContext": "",
          "bContext": "",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.3127,
            0.8318
          ],
          "bRange": [
            0.7961,
            1
          ]
        },
        {
          "metric": "Team knowledge followed, Sonnet vs Haiku: Dreamed notes",
          "aValue": 0.8,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "80% (8/10)",
          "bDisplay": "100% (15/15)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Haiku 4.5 49% to 94%; Claude Sonnet 5.5 80% to 100%), so this sample cannot separate them.",
          "studySlug": "agent-memory",
          "chartId": "memory-team-knowledge-by-model",
          "aN": 10,
          "bN": 15,
          "aContext": "",
          "bContext": "",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.4902,
            0.9433
          ],
          "bRange": [
            0.7961,
            1
          ]
        },
        {
          "metric": "Team knowledge followed, Sonnet vs Haiku: Handbook, 210 lines",
          "aValue": 0.3,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "30% (3/10)",
          "bDisplay": "100% (15/15)",
          "winner": "b",
          "basis": "The 95% intervals do not overlap (Claude Haiku 4.5 11% to 60%; Claude Sonnet 5.5 80% to 100%).",
          "studySlug": "agent-memory",
          "chartId": "memory-team-knowledge-by-model",
          "aN": 10,
          "bN": 15,
          "aContext": "",
          "bContext": "",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.1078,
            0.6032
          ],
          "bRange": [
            0.7961,
            1
          ]
        },
        {
          "metric": "Team knowledge followed, Sonnet vs Haiku: Stop hook only",
          "aValue": 0.8,
          "bValue": 0.6667,
          "unit": "rate",
          "aDisplay": "80% (8/10)",
          "bDisplay": "67% (10/15)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Haiku 4.5 49% to 94%; Claude Sonnet 5.5 42% to 85%), so this sample cannot separate them.",
          "studySlug": "agent-memory",
          "chartId": "memory-team-knowledge-by-model",
          "aN": 10,
          "bN": 15,
          "aContext": "",
          "bContext": "",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.4902,
            0.9433
          ],
          "bRange": [
            0.4171,
            0.8482
          ]
        },
        {
          "metric": "Team knowledge followed, Sonnet vs Haiku: Curated + hook",
          "aValue": 1,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "100% (10/10)",
          "bDisplay": "100% (15/15)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Haiku 4.5 72% to 100%; Claude Sonnet 5.5 80% to 100%), so this sample cannot separate them.",
          "studySlug": "agent-memory",
          "chartId": "memory-team-knowledge-by-model",
          "aN": 10,
          "bN": 15,
          "aContext": "",
          "bContext": "",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.7225,
            1
          ],
          "bRange": [
            0.7961,
            1
          ]
        },
        {
          "metric": "A stale README command: who still ran it?: No memory",
          "aValue": 1,
          "bValue": 0.8,
          "unit": "rate",
          "aDisplay": "100% (10/10)",
          "bDisplay": "80% (12/15)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Haiku 4.5 72% to 100%; Claude Sonnet 5.5 55% to 93%), so this sample cannot separate them.",
          "studySlug": "agent-memory",
          "chartId": "memory-broken-test-command",
          "aN": 10,
          "bN": 15,
          "aContext": "",
          "bContext": "",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.7225,
            1
          ],
          "bRange": [
            0.5481,
            0.9295
          ]
        },
        {
          "metric": "A stale README command: who still ran it?: /init CLAUDE.md",
          "aValue": 1,
          "bValue": 0.8667,
          "unit": "rate",
          "aDisplay": "100% (10/10)",
          "bDisplay": "87% (13/15)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Haiku 4.5 72% to 100%; Claude Sonnet 5.5 62% to 96%), so this sample cannot separate them.",
          "studySlug": "agent-memory",
          "chartId": "memory-broken-test-command",
          "aN": 10,
          "bN": 15,
          "aContext": "",
          "bContext": "",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.7225,
            1
          ],
          "bRange": [
            0.6212,
            0.9626
          ]
        },
        {
          "metric": "A stale README command: who still ran it?: Curated, 11 lines",
          "aValue": 0,
          "bValue": 0,
          "unit": "rate",
          "aDisplay": "0% (0/10)",
          "bDisplay": "0% (0/15)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Haiku 4.5 0% to 28%; Claude Sonnet 5.5 0% to 20%), so this sample cannot separate them.",
          "studySlug": "agent-memory",
          "chartId": "memory-broken-test-command",
          "aN": 10,
          "bN": 15,
          "aContext": "",
          "bContext": "",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0,
            0.2775
          ],
          "bRange": [
            0,
            0.2039
          ]
        },
        {
          "metric": "A stale README command: who still ran it?: Raw notes, 60 lines",
          "aValue": 1,
          "bValue": 0,
          "unit": "rate",
          "aDisplay": "100% (10/10)",
          "bDisplay": "0% (0/15)",
          "winner": "b",
          "basis": "The 95% intervals do not overlap (Claude Haiku 4.5 72% to 100%; Claude Sonnet 5.5 0% to 20%).",
          "studySlug": "agent-memory",
          "chartId": "memory-broken-test-command",
          "aN": 10,
          "bN": 15,
          "aContext": "",
          "bContext": "",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.7225,
            1
          ],
          "bRange": [
            0,
            0.2039
          ]
        },
        {
          "metric": "A stale README command: who still ran it?: Dreamed notes",
          "aValue": 0.1,
          "bValue": 0,
          "unit": "rate",
          "aDisplay": "10% (1/10)",
          "bDisplay": "0% (0/15)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Haiku 4.5 2% to 40%; Claude Sonnet 5.5 0% to 20%), so this sample cannot separate them.",
          "studySlug": "agent-memory",
          "chartId": "memory-broken-test-command",
          "aN": 10,
          "bN": 15,
          "aContext": "",
          "bContext": "",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.0179,
            0.4042
          ],
          "bRange": [
            0,
            0.2039
          ]
        },
        {
          "metric": "A stale README command: who still ran it?: Handbook, 210 lines",
          "aValue": 0,
          "bValue": 0,
          "unit": "rate",
          "aDisplay": "0% (0/10)",
          "bDisplay": "0% (0/15)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Haiku 4.5 0% to 28%; Claude Sonnet 5.5 0% to 20%), so this sample cannot separate them.",
          "studySlug": "agent-memory",
          "chartId": "memory-broken-test-command",
          "aN": 10,
          "bN": 15,
          "aContext": "",
          "bContext": "",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0,
            0.2775
          ],
          "bRange": [
            0,
            0.2039
          ]
        },
        {
          "metric": "A stale README command: who still ran it?: Stop hook only",
          "aValue": 1,
          "bValue": 0.6,
          "unit": "rate",
          "aDisplay": "100% (10/10)",
          "bDisplay": "60% (9/15)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Haiku 4.5 72% to 100%; Claude Sonnet 5.5 36% to 80%), so this sample cannot separate them.",
          "studySlug": "agent-memory",
          "chartId": "memory-broken-test-command",
          "aN": 10,
          "bN": 15,
          "aContext": "",
          "bContext": "",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.7225,
            1
          ],
          "bRange": [
            0.3575,
            0.8018
          ]
        },
        {
          "metric": "A stale README command: who still ran it?: Curated + hook",
          "aValue": 0,
          "bValue": 0,
          "unit": "rate",
          "aDisplay": "0% (0/10)",
          "bDisplay": "0% (0/15)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Haiku 4.5 0% to 28%; Claude Sonnet 5.5 0% to 20%), so this sample cannot separate them.",
          "studySlug": "agent-memory",
          "chartId": "memory-broken-test-command",
          "aN": 10,
          "bN": 15,
          "aContext": "",
          "bContext": "",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0,
            0.2775
          ],
          "bRange": [
            0,
            0.2039
          ]
        },
        {
          "metric": "List-price cost per fully correct result (calculation): No memory",
          "aValue": 0.3786,
          "bValue": 0.1386,
          "unit": "usd",
          "aDisplay": "$0.38",
          "bDisplay": "$0.14",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.38 vs $0.14, 2.7x) is not tested against run-to-run variation.",
          "studySlug": "agent-memory",
          "chartId": "memory-cost-per-full-pass",
          "aN": 2,
          "bN": 9,
          "aContext": "",
          "bContext": "",
          "calculation": true
        },
        {
          "metric": "List-price cost per fully correct result (calculation): /init CLAUDE.md",
          "aValue": 0.4317,
          "bValue": 0.1359,
          "unit": "usd",
          "aDisplay": "$0.43",
          "bDisplay": "$0.14",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.43 vs $0.14, 3.2x) is not tested against run-to-run variation.",
          "studySlug": "agent-memory",
          "chartId": "memory-cost-per-full-pass",
          "aN": 2,
          "bN": 9,
          "aContext": "",
          "bContext": "",
          "calculation": true
        },
        {
          "metric": "List-price cost per fully correct result (calculation): Curated, 11 lines",
          "aValue": 0.1095,
          "bValue": 0.0818,
          "unit": "usd",
          "aDisplay": "$0.11",
          "bDisplay": "$0.082",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.11 vs $0.082) is not tested against run-to-run variation.",
          "studySlug": "agent-memory",
          "chartId": "memory-cost-per-full-pass",
          "aN": 7,
          "bN": 15,
          "aContext": "",
          "bContext": "",
          "calculation": true
        },
        {
          "metric": "List-price cost per fully correct result (calculation): Raw notes, 60 lines",
          "aValue": 0.1255,
          "bValue": 0.1009,
          "unit": "usd",
          "aDisplay": "$0.13",
          "bDisplay": "$0.10",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.13 vs $0.10) is not tested against run-to-run variation.",
          "studySlug": "agent-memory",
          "chartId": "memory-cost-per-full-pass",
          "aN": 6,
          "bN": 14,
          "aContext": "",
          "bContext": "",
          "calculation": true
        },
        {
          "metric": "List-price cost per fully correct result (calculation): Dreamed notes",
          "aValue": 0.1153,
          "bValue": 0.0896,
          "unit": "usd",
          "aDisplay": "$0.12",
          "bDisplay": "$0.090",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.12 vs $0.090) is not tested against run-to-run variation.",
          "studySlug": "agent-memory",
          "chartId": "memory-cost-per-full-pass",
          "aN": 7,
          "bN": 15,
          "aContext": "",
          "bContext": "",
          "calculation": true
        },
        {
          "metric": "List-price cost per fully correct result (calculation): Handbook, 210 lines",
          "aValue": 0.2615,
          "bValue": 0.1006,
          "unit": "usd",
          "aDisplay": "$0.26",
          "bDisplay": "$0.10",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.26 vs $0.10, 2.6x) is not tested against run-to-run variation.",
          "studySlug": "agent-memory",
          "chartId": "memory-cost-per-full-pass",
          "aN": 3,
          "bN": 15,
          "aContext": "",
          "bContext": "",
          "calculation": true
        },
        {
          "metric": "List-price cost per fully correct result (calculation): Stop hook only",
          "aValue": 0.1419,
          "bValue": 0.1278,
          "unit": "usd",
          "aDisplay": "$0.14",
          "bDisplay": "$0.13",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.14 vs $0.13) is not tested against run-to-run variation.",
          "studySlug": "agent-memory",
          "chartId": "memory-cost-per-full-pass",
          "aN": 8,
          "bN": 12,
          "aContext": "",
          "bContext": "",
          "calculation": true
        },
        {
          "metric": "List-price cost per fully correct result (calculation): Curated + hook",
          "aValue": 0.096,
          "bValue": 0.0843,
          "unit": "usd",
          "aDisplay": "$0.096",
          "bDisplay": "$0.084",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.096 vs $0.084) is not tested against run-to-run variation.",
          "studySlug": "agent-memory",
          "chartId": "memory-cost-per-full-pass",
          "aN": 9,
          "bN": 15,
          "aContext": "",
          "bContext": "",
          "calculation": true
        },
        {
          "metric": "Time per session: No memory",
          "aValue": 54.2,
          "bValue": 18,
          "unit": "seconds",
          "aDisplay": "54.2 s",
          "bDisplay": "18.0 s",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap (54.2 s vs 18.0 s, 3.0x) is not tested against run-to-run variation.",
          "studySlug": "agent-memory",
          "chartId": "memory-wall-time",
          "aN": 10,
          "bN": 15,
          "aContext": "",
          "bContext": ""
        },
        {
          "metric": "Time per session: /init CLAUDE.md",
          "aValue": 52.9,
          "bValue": 19,
          "unit": "seconds",
          "aDisplay": "52.9 s",
          "bDisplay": "19.0 s",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap (52.9 s vs 19.0 s, 2.8x) is not tested against run-to-run variation.",
          "studySlug": "agent-memory",
          "chartId": "memory-wall-time",
          "aN": 10,
          "bN": 15,
          "aContext": "",
          "bContext": ""
        },
        {
          "metric": "Time per session: Curated, 11 lines",
          "aValue": 51.7,
          "bValue": 21.9,
          "unit": "seconds",
          "aDisplay": "51.7 s",
          "bDisplay": "21.9 s",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap (51.7 s vs 21.9 s, 2.4x) is not tested against run-to-run variation.",
          "studySlug": "agent-memory",
          "chartId": "memory-wall-time",
          "aN": 10,
          "bN": 15,
          "aContext": "",
          "bContext": ""
        },
        {
          "metric": "Time per session: Raw notes, 60 lines",
          "aValue": 51,
          "bValue": 26.8,
          "unit": "seconds",
          "aDisplay": "51.0 s",
          "bDisplay": "26.8 s",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap (51.0 s vs 26.8 s) is not tested against run-to-run variation.",
          "studySlug": "agent-memory",
          "chartId": "memory-wall-time",
          "aN": 10,
          "bN": 15,
          "aContext": "",
          "bContext": ""
        },
        {
          "metric": "Time per session: Dreamed notes",
          "aValue": 51.8,
          "bValue": 27.2,
          "unit": "seconds",
          "aDisplay": "51.8 s",
          "bDisplay": "27.2 s",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap (51.8 s vs 27.2 s) is not tested against run-to-run variation.",
          "studySlug": "agent-memory",
          "chartId": "memory-wall-time",
          "aN": 10,
          "bN": 15,
          "aContext": "",
          "bContext": ""
        },
        {
          "metric": "Time per session: Handbook, 210 lines",
          "aValue": 49.9,
          "bValue": 23.6,
          "unit": "seconds",
          "aDisplay": "49.9 s",
          "bDisplay": "23.6 s",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap (49.9 s vs 23.6 s, 2.1x) is not tested against run-to-run variation.",
          "studySlug": "agent-memory",
          "chartId": "memory-wall-time",
          "aN": 10,
          "bN": 15,
          "aContext": "",
          "bContext": ""
        },
        {
          "metric": "Time per session: Stop hook only",
          "aValue": 68.5,
          "bValue": 27.3,
          "unit": "seconds",
          "aDisplay": "68.5 s",
          "bDisplay": "27.3 s",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap (68.5 s vs 27.3 s, 2.5x) is not tested against run-to-run variation.",
          "studySlug": "agent-memory",
          "chartId": "memory-wall-time",
          "aN": 10,
          "bN": 15,
          "aContext": "",
          "bContext": ""
        },
        {
          "metric": "Time per session: Curated + hook",
          "aValue": 52.8,
          "bValue": 22,
          "unit": "seconds",
          "aDisplay": "52.8 s",
          "bDisplay": "22.0 s",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap (52.8 s vs 22.0 s, 2.4x) is not tested against run-to-run variation.",
          "studySlug": "agent-memory",
          "chartId": "memory-wall-time",
          "aN": 10,
          "bN": 15,
          "aContext": "",
          "bContext": ""
        },
        {
          "metric": "Typed routing decisions answered exactly right",
          "aValue": 0.8902,
          "bValue": 0.939,
          "unit": "rate",
          "aDisplay": "89% (73/82)",
          "bDisplay": "94% (77/82)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Haiku 4.5 80% to 94%; Claude Sonnet 5.5 87% to 97%), so this sample cannot separate them.",
          "studySlug": "routing-jev-vs-llm",
          "n": 82,
          "chartId": "routing-exact-decisions",
          "aN": 82,
          "bN": 82,
          "aContext": "typed routing decisions · via Claude Code",
          "bContext": "typed routing decisions · via Claude Code",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.8044,
            0.9412
          ],
          "bRange": [
            0.8651,
            0.9737
          ]
        },
        {
          "metric": "Per-question accuracy",
          "aValue": 0.9433,
          "bValue": 0.9742,
          "unit": "rate",
          "aDisplay": "94% (183/194)",
          "bDisplay": "97% (189/194)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Haiku 4.5 90% to 97%; Claude Sonnet 5.5 94% to 99%), so this sample cannot separate them.",
          "studySlug": "routing-jev-vs-llm",
          "n": 194,
          "chartId": "routing-key-accuracy",
          "aN": 194,
          "bN": 194,
          "aContext": "typed routing decisions · via Claude Code",
          "bContext": "typed routing decisions · via Claude Code",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.9013,
            0.968
          ],
          "bRange": [
            0.9411,
            0.9889
          ]
        },
        {
          "metric": "Exact rate by decision type: Failure class",
          "aValue": 0.9444,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "94% (17/18)",
          "bDisplay": "100% (18/18)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Haiku 4.5 74% to 99%; Claude Sonnet 5.5 82% to 100%), so this sample cannot separate them.",
          "studySlug": "routing-jev-vs-llm",
          "n": 18,
          "chartId": "routing-exact-by-decision",
          "aN": 18,
          "bN": 18,
          "aContext": "typed routing decisions · via Claude Code",
          "bContext": "typed routing decisions · via Claude Code",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.7424,
            0.9901
          ],
          "bRange": [
            0.8241,
            1
          ]
        },
        {
          "metric": "Exact rate by decision type: Message intent",
          "aValue": 1,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "100% (20/20)",
          "bDisplay": "100% (20/20)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Haiku 4.5 84% to 100%; Claude Sonnet 5.5 84% to 100%), so this sample cannot separate them.",
          "studySlug": "routing-jev-vs-llm",
          "n": 20,
          "chartId": "routing-exact-by-decision",
          "aN": 20,
          "bN": 20,
          "aContext": "typed routing decisions · via Claude Code",
          "bContext": "typed routing decisions · via Claude Code",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.8389,
            1
          ],
          "bRange": [
            0.8389,
            1
          ]
        },
        {
          "metric": "Exact rate by decision type: Is it a rule?",
          "aValue": 1,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "100% (12/12)",
          "bDisplay": "100% (12/12)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Haiku 4.5 76% to 100%; Claude Sonnet 5.5 76% to 100%), so this sample cannot separate them.",
          "studySlug": "routing-jev-vs-llm",
          "n": 12,
          "chartId": "routing-exact-by-decision",
          "aN": 12,
          "bN": 12,
          "aContext": "typed routing decisions · via Claude Code",
          "bContext": "typed routing decisions · via Claude Code",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.7575,
            1
          ],
          "bRange": [
            0.7575,
            1
          ]
        },
        {
          "metric": "Exact rate by decision type: Context shape",
          "aValue": 0.75,
          "bValue": 0.8438,
          "unit": "rate",
          "aDisplay": "75% (24/32)",
          "bDisplay": "84% (27/32)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Haiku 4.5 58% to 87%; Claude Sonnet 5.5 68% to 93%), so this sample cannot separate them.",
          "studySlug": "routing-jev-vs-llm",
          "n": 32,
          "chartId": "routing-exact-by-decision",
          "aN": 32,
          "bN": 32,
          "aContext": "typed routing decisions · via Claude Code",
          "bContext": "typed routing decisions · via Claude Code",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.5789,
            0.8675
          ],
          "bRange": [
            0.6825,
            0.9314
          ]
        },
        {
          "metric": "Cost per 1,000 routing decisions",
          "aValue": 8.924,
          "bValue": 4.996,
          "unit": "usd",
          "aDisplay": "$8.92",
          "bDisplay": "$5.00",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($8.92 vs $5.00) is not tested against run-to-run variation.",
          "studySlug": "routing-jev-vs-llm",
          "n": 82,
          "chartId": "routing-cost-per-1000",
          "aN": 82,
          "bN": 82,
          "aContext": "typed routing decisions · via Claude Code",
          "bContext": "typed routing decisions · via Claude Code",
          "calculation": true
        },
        {
          "metric": "Time per routing decision (Wall time (CLI))",
          "aValue": 12674,
          "bValue": 2598,
          "unit": "ms",
          "aDisplay": "12,674 ms",
          "bDisplay": "2,598 ms",
          "winner": "b",
          "basis": "Claude Haiku 4.5’s median is above Claude Sonnet 5.5’s 95th percentile (p50–p95 bands: Claude Haiku 4.5 12,674 ms to 34,413 ms; Claude Sonnet 5.5 2,598 ms to 4,298 ms); not a confidence interval.",
          "studySlug": "routing-jev-vs-llm",
          "n": 82,
          "chartId": "routing-decision-latency",
          "aN": 82,
          "bN": 82,
          "aContext": "typed routing decisions · via Claude Code",
          "bContext": "typed routing decisions · via Claude Code",
          "rangeKind": "range",
          "spanKind": "p50-p95",
          "aRange": [
            12674,
            34413
          ],
          "bRange": [
            2598,
            4298
          ]
        },
        {
          "metric": "Time per routing decision (Model time (API))",
          "aValue": 10734,
          "bValue": 1599,
          "unit": "ms",
          "aDisplay": "10,734 ms",
          "bDisplay": "1,599 ms",
          "winner": "b",
          "basis": "Claude Haiku 4.5’s median is above Claude Sonnet 5.5’s 95th percentile (p50–p95 bands: Claude Haiku 4.5 10,734 ms to 32,072 ms; Claude Sonnet 5.5 1,599 ms to 2,574 ms); not a confidence interval.",
          "studySlug": "routing-jev-vs-llm",
          "n": 82,
          "chartId": "routing-decision-latency",
          "aN": 82,
          "bN": 82,
          "aContext": "typed routing decisions · via Claude Code",
          "bContext": "typed routing decisions · via Claude Code",
          "rangeKind": "range",
          "spanKind": "p50-p95",
          "aRange": [
            10734,
            32072
          ],
          "bRange": [
            1599,
            2574
          ]
        },
        {
          "metric": "Time to make one routing decision",
          "aValue": 12543,
          "bValue": 2597,
          "unit": "ms",
          "aDisplay": "12,543 ms",
          "bDisplay": "2,597 ms",
          "winner": "b",
          "basis": "Claude Haiku 4.5’s median is above Claude Sonnet 5.5’s 95th percentile (p50–p95 bands: Claude Haiku 4.5 12,543 ms to 34,481 ms; Claude Sonnet 5.5 2,597 ms to 4,298 ms); not a confidence interval.",
          "studySlug": "routing-overhead",
          "n": 82,
          "chartId": "router-overhead-decision-latency",
          "aN": 82,
          "bN": 82,
          "aContext": "thinking on · via Claude Code · routing overhead per decision",
          "bContext": "effort low · via Claude Code · routing overhead per decision",
          "rangeKind": "range",
          "spanKind": "p50-p95",
          "aRange": [
            12543,
            34481
          ],
          "bRange": [
            2597,
            4298
          ]
        },
        {
          "metric": "Where an LLM router’s time goes: model vs CLI (Model API time)",
          "aValue": 10508,
          "bValue": 1596,
          "unit": "ms",
          "aDisplay": "10,508 ms",
          "bDisplay": "1,596 ms",
          "winner": "b",
          "basis": "Claude Haiku 4.5’s median is above Claude Sonnet 5.5’s 95th percentile (p50–p95 bands: Claude Haiku 4.5 10,508 ms to 32,132 ms; Claude Sonnet 5.5 1,596 ms to 2,583 ms); not a confidence interval.",
          "studySlug": "routing-overhead",
          "n": 82,
          "chartId": "router-overhead-cli-vs-model-time",
          "aN": 82,
          "bN": 82,
          "aContext": "thinking on · via Claude Code · routing overhead per decision",
          "bContext": "effort low · via Claude Code · routing overhead per decision",
          "rangeKind": "range",
          "spanKind": "p50-p95",
          "aRange": [
            10508,
            32132
          ],
          "bRange": [
            1596,
            2583
          ]
        },
        {
          "metric": "Where an LLM router’s time goes: model vs CLI (CLI and harness time)",
          "aValue": 1698,
          "bValue": 973,
          "unit": "ms",
          "aDisplay": "1,698 ms",
          "bDisplay": "973 ms",
          "winner": "b",
          "basis": "Claude Haiku 4.5’s median is above Claude Sonnet 5.5’s 95th percentile (p50–p95 bands: Claude Haiku 4.5 1,698 ms to 2,677 ms; Claude Sonnet 5.5 973 ms to 1,277 ms); not a confidence interval.",
          "studySlug": "routing-overhead",
          "n": 82,
          "chartId": "router-overhead-cli-vs-model-time",
          "aN": 82,
          "bN": 82,
          "aContext": "thinking on · via Claude Code · routing overhead per decision",
          "bContext": "effort low · via Claude Code · routing overhead per decision",
          "rangeKind": "range",
          "spanKind": "p50-p95",
          "aRange": [
            1698,
            2677
          ],
          "bRange": [
            973,
            1277
          ]
        },
        {
          "metric": "Routing calls that returned a decision",
          "aValue": 1,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "100% (82/82)",
          "bDisplay": "100% (82/82)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Haiku 4.5 96% to 100%; Claude Sonnet 5.5 96% to 100%), so this sample cannot separate them.",
          "studySlug": "routing-overhead",
          "n": 82,
          "chartId": "router-overhead-completed",
          "aN": 82,
          "bN": 82,
          "aContext": "thinking on · via Claude Code · routing overhead per decision",
          "bContext": "effort low · via Claude Code · routing overhead per decision",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.9552,
            1
          ],
          "bRange": [
            0.9552,
            1
          ]
        },
        {
          "metric": "Added routing cost per 1,000 tasks (calculation) (Every model call routed (49.5 per task))",
          "aValue": 441.74,
          "bValue": 247.3,
          "unit": "usd",
          "aDisplay": "$441.74",
          "bDisplay": "$247.30",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($441.74 vs $247.30) is not tested against run-to-run variation.",
          "studySlug": "routing-overhead",
          "chartId": "router-overhead-cost-per-1000-tasks",
          "aContext": "thinking on · via Claude Code · calculation per 1,000 tasks from recorded decision counts",
          "bContext": "effort low · via Claude Code · calculation per 1,000 tasks from recorded decision counts",
          "calculation": true
        },
        {
          "metric": "Added routing cost per 1,000 tasks (calculation) (Only System One decisions (7 per task))",
          "aValue": 62.47,
          "bValue": 34.97,
          "unit": "usd",
          "aDisplay": "$62.47",
          "bDisplay": "$34.97",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($62.47 vs $34.97) is not tested against run-to-run variation.",
          "studySlug": "routing-overhead",
          "chartId": "router-overhead-cost-per-1000-tasks",
          "aContext": "thinking on · via Claude Code · calculation per 1,000 tasks from recorded decision counts",
          "bContext": "effort low · via Claude Code · calculation per 1,000 tasks from recorded decision counts",
          "calculation": true
        },
        {
          "metric": "Added routing delay per task (calculation) (Every model call routed (49.5 per task))",
          "aValue": 620.8785,
          "bValue": 128.5515,
          "unit": "seconds",
          "aDisplay": "620.9 s",
          "bDisplay": "128.6 s",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap (620.9 s vs 128.6 s, 4.8x) is not tested against run-to-run variation.",
          "studySlug": "routing-overhead",
          "chartId": "router-overhead-delay-per-task",
          "aContext": "thinking on · via Claude Code · calculation per task from recorded decision counts, decisions in line",
          "bContext": "effort low · via Claude Code · calculation per task from recorded decision counts, decisions in line",
          "calculation": true
        },
        {
          "metric": "Added routing delay per task (calculation) (Only System One decisions (7 per task))",
          "aValue": 87.801,
          "bValue": 18.179,
          "unit": "seconds",
          "aDisplay": "87.8 s",
          "bDisplay": "18.2 s",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap (87.8 s vs 18.2 s, 4.8x) is not tested against run-to-run variation.",
          "studySlug": "routing-overhead",
          "chartId": "router-overhead-delay-per-task",
          "aContext": "thinking on · via Claude Code · calculation per task from recorded decision counts, decisions in line",
          "bContext": "effort low · via Claude Code · calculation per task from recorded decision counts, decisions in line",
          "calculation": true
        },
        {
          "metric": "Strict pass rate: single call vs agent loop on eight hard tasks",
          "aValue": 0.4583,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "46% (11/24)",
          "bDisplay": "100% (24/24)",
          "winner": "b",
          "basis": "The 95% intervals do not overlap (Claude Haiku 4.5 28% to 65%; Claude Sonnet 5.5 86% to 100%).",
          "studySlug": "single-call-vs-agent-loop",
          "n": 24,
          "chartId": "agent-loop-pass-rate",
          "aN": 24,
          "bN": 24,
          "aContext": "Claude Code · single call",
          "bContext": "Claude Code · single call",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.2789,
            0.6493
          ],
          "bRange": [
            0.862,
            1
          ]
        },
        {
          "metric": "Strict passes per task: single call vs agent loop: Interval merge fix",
          "aValue": 1,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "100% (3/3)",
          "bDisplay": "100% (3/3)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Haiku 4.5 44% to 100%; Claude Sonnet 5.5 44% to 100%), so this sample cannot separate them.",
          "studySlug": "single-call-vs-agent-loop",
          "n": 3,
          "chartId": "agent-loop-by-task",
          "aN": 3,
          "bN": 3,
          "aContext": "Claude Code · single call",
          "bContext": "Claude Code · single call",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.4385,
            1
          ],
          "bRange": [
            0.4385,
            1
          ]
        },
        {
          "metric": "Strict passes per task: single call vs agent loop: DST day-length fix",
          "aValue": 0.3333,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "33% (1/3)",
          "bDisplay": "100% (3/3)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Haiku 4.5 6% to 79%; Claude Sonnet 5.5 44% to 100%), so this sample cannot separate them.",
          "studySlug": "single-call-vs-agent-loop",
          "n": 3,
          "chartId": "agent-loop-by-task",
          "aN": 3,
          "bN": 3,
          "aContext": "Claude Code · single call",
          "bContext": "Claude Code · single call",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.0615,
            0.7923
          ],
          "bRange": [
            0.4385,
            1
          ]
        },
        {
          "metric": "Strict passes per task: single call vs agent loop: CSV parser",
          "aValue": 0.6667,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "67% (2/3)",
          "bDisplay": "100% (3/3)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Haiku 4.5 21% to 94%; Claude Sonnet 5.5 44% to 100%), so this sample cannot separate them.",
          "studySlug": "single-call-vs-agent-loop",
          "n": 3,
          "chartId": "agent-loop-by-task",
          "aN": 3,
          "bN": 3,
          "aContext": "Claude Code · single call",
          "bContext": "Claude Code · single call",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.2077,
            0.9385
          ],
          "bRange": [
            0.4385,
            1
          ]
        },
        {
          "metric": "Strict passes per task: single call vs agent loop: Event-loop order",
          "aValue": 0,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "0% (0/3)",
          "bDisplay": "100% (3/3)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Haiku 4.5 0% to 56%; Claude Sonnet 5.5 44% to 100%), so this sample cannot separate them.",
          "studySlug": "single-call-vs-agent-loop",
          "n": 3,
          "chartId": "agent-loop-by-task",
          "aN": 3,
          "bN": 3,
          "aContext": "Claude Code · single call",
          "bContext": "Claude Code · single call",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0,
            0.5615
          ],
          "bRange": [
            0.4385,
            1
          ]
        },
        {
          "metric": "Strict passes per task: single call vs agent loop: Room schedule",
          "aValue": 0,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "0% (0/3)",
          "bDisplay": "100% (3/3)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Haiku 4.5 0% to 56%; Claude Sonnet 5.5 44% to 100%), so this sample cannot separate them.",
          "studySlug": "single-call-vs-agent-loop",
          "n": 3,
          "chartId": "agent-loop-by-task",
          "aN": 3,
          "bN": 3,
          "aContext": "Claude Code · single call",
          "bContext": "Claude Code · single call",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0,
            0.5615
          ],
          "bRange": [
            0.4385,
            1
          ]
        },
        {
          "metric": "Strict passes per task: single call vs agent loop: SemVer regex",
          "aValue": 1,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "100% (3/3)",
          "bDisplay": "100% (3/3)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Haiku 4.5 44% to 100%; Claude Sonnet 5.5 44% to 100%), so this sample cannot separate them.",
          "studySlug": "single-call-vs-agent-loop",
          "n": 3,
          "chartId": "agent-loop-by-task",
          "aN": 3,
          "bN": 3,
          "aContext": "Claude Code · single call",
          "bContext": "Claude Code · single call",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.4385,
            1
          ],
          "bRange": [
            0.4385,
            1
          ]
        },
        {
          "metric": "Strict passes per task: single call vs agent loop: Money refactor",
          "aValue": 0.6667,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "67% (2/3)",
          "bDisplay": "100% (3/3)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Haiku 4.5 21% to 94%; Claude Sonnet 5.5 44% to 100%), so this sample cannot separate them.",
          "studySlug": "single-call-vs-agent-loop",
          "n": 3,
          "chartId": "agent-loop-by-task",
          "aN": 3,
          "bN": 3,
          "aContext": "Claude Code · single call",
          "bContext": "Claude Code · single call",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.2077,
            0.9385
          ],
          "bRange": [
            0.4385,
            1
          ]
        },
        {
          "metric": "Strict passes per task: single call vs agent loop: SQL report",
          "aValue": 0,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "0% (0/3)",
          "bDisplay": "100% (3/3)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Haiku 4.5 0% to 56%; Claude Sonnet 5.5 44% to 100%), so this sample cannot separate them.",
          "studySlug": "single-call-vs-agent-loop",
          "n": 3,
          "chartId": "agent-loop-by-task",
          "aN": 3,
          "bN": 3,
          "aContext": "Claude Code · single call",
          "bContext": "Claude Code · single call",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0,
            0.5615
          ],
          "bRange": [
            0.4385,
            1
          ]
        },
        {
          "metric": "Total time per attempt: single call vs agent loop",
          "aValue": 39.01,
          "bValue": 7.75,
          "unit": "seconds",
          "aDisplay": "39.0 s",
          "bDisplay": "7.75 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Haiku 4.5 15.3 s to 75.1 s; Claude Sonnet 5.5 2.26 s to 34.8 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "single-call-vs-agent-loop",
          "n": 24,
          "chartId": "agent-loop-total-time",
          "aN": 24,
          "bN": 24,
          "aContext": "Claude Code · single call",
          "bContext": "Claude Code · single call",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            15.27,
            75.13
          ],
          "bRange": [
            2.26,
            34.79
          ]
        },
        {
          "metric": "Tokens per attempt: single call vs agent loop (Input tokens (cache reads included))",
          "aValue": 3941,
          "bValue": 2281,
          "unit": "tokens",
          "aDisplay": "3,941",
          "bDisplay": "2,281",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "single-call-vs-agent-loop",
          "n": 24,
          "chartId": "agent-loop-tokens",
          "aN": 24,
          "bN": 24,
          "aContext": "Claude Code · single call",
          "bContext": "Claude Code · single call",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            3879,
            4221
          ],
          "bRange": [
            2234,
            2669
          ]
        },
        {
          "metric": "Tokens per attempt: single call vs agent loop (Output tokens)",
          "aValue": 5064,
          "bValue": 1050,
          "unit": "tokens",
          "aDisplay": "5,064",
          "bDisplay": "1,050",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "single-call-vs-agent-loop",
          "n": 24,
          "chartId": "agent-loop-tokens",
          "aN": 24,
          "bN": 24,
          "aContext": "Claude Code · single call",
          "bContext": "Claude Code · single call",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            1899,
            9321
          ],
          "bRange": [
            176,
            3895
          ]
        },
        {
          "metric": "Tool calls per agent-loop attempt",
          "aValue": 3,
          "bValue": 0,
          "unit": "count",
          "aDisplay": "3",
          "bDisplay": "0",
          "winner": "unclear",
          "basis": "More or fewer count is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-tool-calls",
          "aN": 24,
          "bN": 16,
          "aContext": "Claude Code · agent loop",
          "bContext": "Claude Code · agent loop",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            2,
            18
          ],
          "bRange": [
            0,
            3
          ]
        },
        {
          "metric": "List-price cost per strict pass: single call vs agent loop (calculation)",
          "aValue": 0.0672,
          "bValue": 0.01435,
          "unit": "usd",
          "aDisplay": "$0.067",
          "bDisplay": "$0.014",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.067 vs $0.014, 4.7x) is not tested against run-to-run variation.",
          "studySlug": "single-call-vs-agent-loop",
          "n": 24,
          "chartId": "agent-loop-cost-per-pass",
          "aN": 24,
          "bN": 24,
          "aContext": "Claude Code · single call",
          "bContext": "Claude Code · single call",
          "calculation": true
        },
        {
          "metric": "Haiku thinking study: typed routing decisions answered exactly right (Exact decisions (every scored question right))",
          "aValue": 0.8659,
          "bValue": 0.939,
          "unit": "rate",
          "aDisplay": "87% (71/82)",
          "bDisplay": "94% (77/82)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Haiku 4.5 78% to 92%; Claude Sonnet 5.5 87% to 97%), so this sample cannot separate them.",
          "studySlug": "haiku-thinking-on-off",
          "n": 82,
          "chartId": "haiku-thinking-router-exact",
          "aN": 82,
          "bN": 82,
          "aContext": "Claude Code · thinking off · typed routing decisions, thinking on vs off",
          "bContext": "Claude Code · effort low · typed routing decisions, thinking on vs off",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.7755,
            0.9234
          ],
          "bRange": [
            0.8651,
            0.9737
          ]
        },
        {
          "metric": "Haiku thinking study: typed routing decisions answered exactly right (Per-question accuracy)",
          "aValue": 0.9124,
          "bValue": 0.9742,
          "unit": "rate",
          "aDisplay": "91% (177/194)",
          "bDisplay": "97% (189/194)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Haiku 4.5 86% to 94%; Claude Sonnet 5.5 94% to 99%), so this sample cannot separate them.",
          "studySlug": "haiku-thinking-on-off",
          "n": 194,
          "chartId": "haiku-thinking-router-exact",
          "aN": 194,
          "bN": 194,
          "aContext": "Claude Code · thinking off · typed routing decisions, thinking on vs off",
          "bContext": "Claude Code · effort low · typed routing decisions, thinking on vs off",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.8642,
            0.9446
          ],
          "bRange": [
            0.9411,
            0.9889
          ]
        },
        {
          "metric": "Haiku thinking study: time per routing decision (Wall time (CLI))",
          "aValue": 4.66,
          "bValue": 2.6,
          "unit": "seconds",
          "aDisplay": "4.66 s",
          "bDisplay": "2.60 s",
          "winner": "b",
          "basis": "Claude Haiku 4.5’s median is above Claude Sonnet 5.5’s 95th percentile (p50–p95 bands: Claude Haiku 4.5 4.66 s to 8.18 s; Claude Sonnet 5.5 2.60 s to 4.30 s); not a confidence interval.",
          "studySlug": "haiku-thinking-on-off",
          "n": 82,
          "chartId": "haiku-thinking-router-latency",
          "aN": 82,
          "bN": 82,
          "aContext": "Claude Code · thinking off · typed routing decisions, thinking on vs off",
          "bContext": "Claude Code · effort low · typed routing decisions, thinking on vs off",
          "rangeKind": "range",
          "spanKind": "p50-p95",
          "aRange": [
            4.66,
            8.18
          ],
          "bRange": [
            2.6,
            4.3
          ]
        },
        {
          "metric": "Haiku thinking study: time per routing decision (Model time (API))",
          "aValue": 3.79,
          "bValue": 1.6,
          "unit": "seconds",
          "aDisplay": "3.79 s",
          "bDisplay": "1.60 s",
          "winner": "b",
          "basis": "Claude Haiku 4.5’s median is above Claude Sonnet 5.5’s 95th percentile (p50–p95 bands: Claude Haiku 4.5 3.79 s to 7.43 s; Claude Sonnet 5.5 1.60 s to 2.58 s); not a confidence interval.",
          "studySlug": "haiku-thinking-on-off",
          "n": 82,
          "chartId": "haiku-thinking-router-latency",
          "aN": 82,
          "bN": 82,
          "aContext": "Claude Code · thinking off · typed routing decisions, thinking on vs off",
          "bContext": "Claude Code · effort low · typed routing decisions, thinking on vs off",
          "rangeKind": "range",
          "spanKind": "p50-p95",
          "aRange": [
            3.79,
            7.43
          ],
          "bRange": [
            1.6,
            2.58
          ]
        },
        {
          "metric": "Haiku thinking study: thinking and visible output tokens per routing decision (Thinking tokens)",
          "aValue": 0,
          "bValue": 2,
          "unit": "tokens",
          "aDisplay": "0",
          "bDisplay": "2",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "haiku-thinking-on-off",
          "n": 82,
          "chartId": "haiku-thinking-router-tokens",
          "aN": 82,
          "bN": 82,
          "aContext": "Claude Code · thinking off · typed routing decisions, thinking on vs off",
          "bContext": "Claude Code · effort low · typed routing decisions, thinking on vs off"
        },
        {
          "metric": "Haiku thinking study: thinking and visible output tokens per routing decision (Visible output tokens)",
          "aValue": 366,
          "bValue": 105,
          "unit": "tokens",
          "aDisplay": "366",
          "bDisplay": "105",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "haiku-thinking-on-off",
          "n": 82,
          "chartId": "haiku-thinking-router-tokens",
          "aN": 82,
          "bN": 82,
          "aContext": "Claude Code · thinking off · typed routing decisions, thinking on vs off",
          "bContext": "Claude Code · effort low · typed routing decisions, thinking on vs off"
        },
        {
          "metric": "Haiku thinking study: list-price cost per 1,000 routing decisions (calculation)",
          "aValue": 3.364,
          "bValue": 7.324,
          "unit": "usd",
          "aDisplay": "$3.36",
          "bDisplay": "$7.32",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($3.36 vs $7.32, 2.2x) is not tested against run-to-run variation.",
          "studySlug": "haiku-thinking-on-off",
          "n": 82,
          "chartId": "haiku-thinking-router-cost",
          "aN": 82,
          "bN": 82,
          "aContext": "Claude Code · thinking off · typed routing decisions, thinking on vs off",
          "bContext": "Claude Code · effort low · typed routing decisions, thinking on vs off",
          "calculation": true
        },
        {
          "metric": "Does a JSON schema raise the pass rate? Instructions vs schema mode (Strict pass: the whole reply is the right JSON)",
          "aValue": 0,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "0% (0/24)",
          "bDisplay": "100% (12/12)",
          "winner": "b",
          "basis": "The 95% intervals do not overlap (Claude Haiku 4.5 0% to 14%; Claude Sonnet 5.5 76% to 100%).",
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-pass-rate",
          "aN": 24,
          "bN": 12,
          "aContext": "Claude Code · instructions",
          "bContext": "Claude Code · instructions",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0,
            0.138
          ],
          "bRange": [
            0.7575,
            1
          ],
          "calculation": true
        },
        {
          "metric": "Does a JSON schema raise the pass rate? Instructions vs schema mode (Right answer in any format (strict pass or format miss))",
          "aValue": 0.7083,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "71% (17/24)",
          "bDisplay": "100% (12/12)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Haiku 4.5 51% to 85%; Claude Sonnet 5.5 76% to 100%), so this sample cannot separate them.",
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-pass-rate",
          "aN": 24,
          "bN": 12,
          "aContext": "Claude Code · instructions",
          "bContext": "Claude Code · instructions",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.5083,
            0.8509
          ],
          "bRange": [
            0.7575,
            1
          ],
          "calculation": true
        },
        {
          "metric": "What each call produced: strict pass, format miss, wrong values or error (Strict pass)",
          "aValue": 0,
          "bValue": 12,
          "unit": "count",
          "aDisplay": "0",
          "bDisplay": "12",
          "winner": "unclear",
          "basis": "More or fewer count is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-outcomes",
          "aN": 24,
          "bN": 12,
          "aContext": "Claude Code · instructions",
          "bContext": "Claude Code · instructions"
        },
        {
          "metric": "What each call produced: strict pass, format miss, wrong values or error (Format miss)",
          "aValue": 17,
          "bValue": 0,
          "unit": "count",
          "aDisplay": "17",
          "bDisplay": "0",
          "winner": "unclear",
          "basis": "More or fewer count is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-outcomes",
          "aN": 24,
          "bN": 12,
          "aContext": "Claude Code · instructions",
          "bContext": "Claude Code · instructions"
        },
        {
          "metric": "What each call produced: strict pass, format miss, wrong values or error (Wrong values)",
          "aValue": 7,
          "bValue": 0,
          "unit": "count",
          "aDisplay": "7",
          "bDisplay": "0",
          "winner": "unclear",
          "basis": "More or fewer count is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-outcomes",
          "aN": 24,
          "bN": 12,
          "aContext": "Claude Code · instructions",
          "bContext": "Claude Code · instructions"
        },
        {
          "metric": "What each call produced: strict pass, format miss, wrong values or error (Error)",
          "aValue": 0,
          "bValue": 0,
          "unit": "count",
          "aDisplay": "0",
          "bDisplay": "0",
          "winner": "tie",
          "basis": "Same value. More or fewer is not better by itself for this metric.",
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-outcomes",
          "aN": 24,
          "bN": 12,
          "aContext": "Claude Code · instructions",
          "bContext": "Claude Code · instructions"
        },
        {
          "metric": "Time per call, instructions vs schema mode",
          "aValue": 9.52,
          "bValue": 3.52,
          "unit": "seconds",
          "aDisplay": "9.52 s",
          "bDisplay": "3.52 s",
          "winner": "b",
          "basis": "The run ranges (fastest to slowest) do not overlap (Claude Haiku 4.5 5.67 s to 17.0 s; Claude Sonnet 5.5 2.67 s to 4.12 s). A range is not a confidence interval.",
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-time",
          "aN": 24,
          "bN": 12,
          "aContext": "Claude Code · instructions",
          "bContext": "Claude Code · instructions",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            5.67,
            17
          ],
          "bRange": [
            2.67,
            4.12
          ]
        },
        {
          "metric": "Output and reasoning tokens per call, instructions vs schema mode (Median output tokens per call)",
          "aValue": 1128,
          "bValue": 368,
          "unit": "tokens",
          "aDisplay": "1,128",
          "bDisplay": "368",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-tokens",
          "aN": 24,
          "bN": 12,
          "aContext": "Claude Code · instructions",
          "bContext": "Claude Code · instructions"
        },
        {
          "metric": "Unseen routing decisions answered exactly right",
          "aValue": 0.7857,
          "bValue": 0.875,
          "unit": "rate",
          "aDisplay": "79% (44/56)",
          "bDisplay": "88% (49/56)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Haiku 4.5 66% to 87%; Claude Sonnet 5.5 76% to 94%), so this sample cannot separate them.",
          "studySlug": "routing-holdout",
          "n": 56,
          "chartId": "routing-holdout-exact",
          "aN": 56,
          "bN": 56,
          "aContext": "Claude Code",
          "bContext": "Claude Code · effort low",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.6618,
            0.8729
          ],
          "bRange": [
            0.7637,
            0.9381
          ]
        },
        {
          "metric": "Per-question accuracy on unseen decisions",
          "aValue": 0.816,
          "bValue": 0.92,
          "unit": "rate",
          "aDisplay": "82% (102/125)",
          "bDisplay": "92% (115/125)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Haiku 4.5 74% to 87%; Claude Sonnet 5.5 86% to 96%), so this sample cannot separate them.",
          "studySlug": "routing-holdout",
          "n": 125,
          "chartId": "routing-holdout-key-accuracy",
          "aN": 125,
          "bN": 125,
          "aContext": "Claude Code",
          "bContext": "Claude Code · effort low",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.739,
            0.8741
          ],
          "bRange": [
            0.859,
            0.956
          ]
        },
        {
          "metric": "Exact rate on unseen decisions, by decision type: Failure class",
          "aValue": 0.9286,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "93% (13/14)",
          "bDisplay": "100% (14/14)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Haiku 4.5 69% to 99%; Claude Sonnet 5.5 78% to 100%), so this sample cannot separate them.",
          "studySlug": "routing-holdout",
          "n": 14,
          "chartId": "routing-holdout-by-purpose",
          "aN": 14,
          "bN": 14,
          "aContext": "Claude Code",
          "bContext": "Claude Code · effort low",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.6853,
            0.9873
          ],
          "bRange": [
            0.7847,
            1
          ]
        },
        {
          "metric": "Exact rate on unseen decisions, by decision type: Message intent",
          "aValue": 0.9286,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "93% (13/14)",
          "bDisplay": "100% (14/14)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Haiku 4.5 69% to 99%; Claude Sonnet 5.5 78% to 100%), so this sample cannot separate them.",
          "studySlug": "routing-holdout",
          "n": 14,
          "chartId": "routing-holdout-by-purpose",
          "aN": 14,
          "bN": 14,
          "aContext": "Claude Code",
          "bContext": "Claude Code · effort low",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.6853,
            0.9873
          ],
          "bRange": [
            0.7847,
            1
          ]
        },
        {
          "metric": "Exact rate on unseen decisions, by decision type: Is it a rule?",
          "aValue": 0.9286,
          "bValue": 0.9286,
          "unit": "rate",
          "aDisplay": "93% (13/14)",
          "bDisplay": "93% (13/14)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Haiku 4.5 69% to 99%; Claude Sonnet 5.5 69% to 99%), so this sample cannot separate them.",
          "studySlug": "routing-holdout",
          "n": 14,
          "chartId": "routing-holdout-by-purpose",
          "aN": 14,
          "bN": 14,
          "aContext": "Claude Code",
          "bContext": "Claude Code · effort low",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.6853,
            0.9873
          ],
          "bRange": [
            0.6853,
            0.9873
          ]
        },
        {
          "metric": "Exact rate on unseen decisions, by decision type: Context shape",
          "aValue": 0.3571,
          "bValue": 0.5714,
          "unit": "rate",
          "aDisplay": "36% (5/14)",
          "bDisplay": "57% (8/14)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Haiku 4.5 16% to 61%; Claude Sonnet 5.5 33% to 79%), so this sample cannot separate them.",
          "studySlug": "routing-holdout",
          "n": 14,
          "chartId": "routing-holdout-by-purpose",
          "aN": 14,
          "bN": 14,
          "aContext": "Claude Code",
          "bContext": "Claude Code · effort low",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.1634,
            0.6124
          ],
          "bRange": [
            0.3259,
            0.7862
          ]
        },
        {
          "metric": "Tuned case set vs unseen holdout: exact rate per router (Tuned set (routing-jev-vs-llm))",
          "aValue": 0.8902,
          "bValue": 0.939,
          "unit": "rate",
          "aDisplay": "89% (73/82)",
          "bDisplay": "94% (77/82)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Haiku 4.5 80% to 94%; Claude Sonnet 5.5 87% to 97%), so this sample cannot separate them.",
          "studySlug": "routing-holdout",
          "n": 82,
          "chartId": "routing-holdout-tuned-vs-unseen",
          "aN": 82,
          "bN": 82,
          "aContext": "Claude Code",
          "bContext": "Claude Code · effort low",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.8044,
            0.9412
          ],
          "bRange": [
            0.8651,
            0.9737
          ]
        },
        {
          "metric": "Tuned case set vs unseen holdout: exact rate per router (Unseen holdout)",
          "aValue": 0.7857,
          "bValue": 0.875,
          "unit": "rate",
          "aDisplay": "79% (44/56)",
          "bDisplay": "88% (49/56)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Haiku 4.5 66% to 87%; Claude Sonnet 5.5 76% to 94%), so this sample cannot separate them.",
          "studySlug": "routing-holdout",
          "n": 56,
          "chartId": "routing-holdout-tuned-vs-unseen",
          "aN": 56,
          "bN": 56,
          "aContext": "Claude Code",
          "bContext": "Claude Code · effort low",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.6618,
            0.8729
          ],
          "bRange": [
            0.7637,
            0.9381
          ]
        },
        {
          "metric": "Time per routing decision, by route (Wall time)",
          "aValue": 9.444,
          "bValue": 2.359,
          "unit": "seconds",
          "aDisplay": "9.44 s",
          "bDisplay": "2.36 s",
          "winner": "b",
          "basis": "Claude Haiku 4.5’s median is above Claude Sonnet 5.5’s 95th percentile (p50–p95 bands: Claude Haiku 4.5 9.44 s to 25.4 s; Claude Sonnet 5.5 2.36 s to 3.66 s); not a confidence interval.",
          "studySlug": "routing-holdout",
          "n": 56,
          "chartId": "routing-holdout-latency",
          "aN": 56,
          "bN": 56,
          "aContext": "Claude Code",
          "bContext": "Claude Code · effort low",
          "rangeKind": "range",
          "spanKind": "p50-p95",
          "aRange": [
            9.444,
            25.413
          ],
          "bRange": [
            2.359,
            3.657
          ]
        },
        {
          "metric": "Time per routing decision, by route (Model time (API, CLI-reported))",
          "aValue": 7.522,
          "bValue": 1.485,
          "unit": "seconds",
          "aDisplay": "7.52 s",
          "bDisplay": "1.49 s",
          "winner": "b",
          "basis": "Claude Haiku 4.5’s median is above Claude Sonnet 5.5’s 95th percentile (p50–p95 bands: Claude Haiku 4.5 7.52 s to 23.9 s; Claude Sonnet 5.5 1.49 s to 2.38 s); not a confidence interval.",
          "studySlug": "routing-holdout",
          "n": 56,
          "chartId": "routing-holdout-latency",
          "aN": 56,
          "bN": 56,
          "aContext": "Claude Code",
          "bContext": "Claude Code · effort low",
          "rangeKind": "range",
          "spanKind": "p50-p95",
          "aRange": [
            7.522,
            23.913
          ],
          "bRange": [
            1.485,
            2.377
          ]
        },
        {
          "metric": "Cost per 1,000 unseen routing decisions",
          "aValue": 7.129,
          "bValue": 7.244,
          "unit": "usd",
          "aDisplay": "$7.13",
          "bDisplay": "$7.24",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($7.13 vs $7.24) is not tested against run-to-run variation.",
          "studySlug": "routing-holdout",
          "n": 56,
          "chartId": "routing-holdout-cost-per-1000",
          "aN": 56,
          "bN": 56,
          "aContext": "Claude Code",
          "bContext": "Claude Code · effort low",
          "calculation": true
        },
        {
          "metric": "Reasoning share of output tokens per call on hard tasks (calculation)",
          "aValue": 91.68,
          "bValue": 54.54,
          "unit": "percent",
          "aDisplay": "91.7%",
          "bDisplay": "54.5%",
          "winner": "unclear",
          "basis": "More or fewer percent is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "n": 24,
          "chartId": "thinking-bill-share",
          "aN": 24,
          "bN": 24,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            76.46,
            99.27
          ],
          "bRange": [
            0,
            95.91
          ],
          "calculation": true
        },
        {
          "metric": "List-price cost per call: reasoning, remaining output and input (calculation) (Reasoning (output tokens))",
          "aValue": 0.024492,
          "bValue": 0.006665,
          "unit": "usd",
          "aDisplay": "$0.024",
          "bDisplay": "$0.0067",
          "winner": "unclear",
          "basis": "More or fewer usd is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "n": 24,
          "chartId": "thinking-bill-cost-per-call",
          "aN": 24,
          "bN": 24,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "calculation": true
        },
        {
          "metric": "List-price cost per call: reasoning, remaining output and input (calculation) (Remaining output (visible-answer estimate))",
          "aValue": 0.001795,
          "bValue": 0.003672,
          "unit": "usd",
          "aDisplay": "$0.0018",
          "bDisplay": "$0.0037",
          "winner": "unclear",
          "basis": "More or fewer usd is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "n": 24,
          "chartId": "thinking-bill-cost-per-call",
          "aN": 24,
          "bN": 24,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "calculation": true
        },
        {
          "metric": "List-price cost per call: reasoning, remaining output and input (calculation) (Input (prompt, cache priced))",
          "aValue": 0.00451,
          "bValue": 0.004012,
          "unit": "usd",
          "aDisplay": "$0.0045",
          "bDisplay": "$0.0040",
          "winner": "unclear",
          "basis": "More or fewer usd is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "n": 24,
          "chartId": "thinking-bill-cost-per-call",
          "aN": 24,
          "bN": 24,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "calculation": true
        },
        {
          "metric": "Reasoning share on short tasks vs hard tasks (calculation) (Eight hard tasks)",
          "aValue": 91.68,
          "bValue": 54.54,
          "unit": "percent",
          "aDisplay": "91.7%",
          "bDisplay": "54.5%",
          "winner": "unclear",
          "basis": "More or fewer percent is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "n": 24,
          "chartId": "thinking-bill-short-vs-hard",
          "aN": 24,
          "bN": 24,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            76.46,
            99.27
          ],
          "bRange": [
            0,
            95.91
          ],
          "calculation": true
        },
        {
          "metric": "Reasoning share on short tasks vs hard tasks (calculation) (Five short tasks)",
          "aValue": 90.19,
          "bValue": 0,
          "unit": "percent",
          "aDisplay": "90.2%",
          "bDisplay": "0%",
          "winner": "unclear",
          "basis": "More or fewer percent is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "n": 15,
          "chartId": "thinking-bill-short-vs-hard",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            73.1,
            97.59
          ],
          "bRange": [
            0,
            72.75
          ],
          "calculation": true
        },
        {
          "metric": "Time to first text: a 250-line answer, six models",
          "aValue": 4,
          "bValue": 1.96,
          "unit": "seconds",
          "aDisplay": "4.00 s",
          "bDisplay": "1.96 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Haiku 4.5 2.84 s to 6.38 s; Claude Sonnet 5.5 0.88 s to 4.09 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "llm-speed-anatomy",
          "n": 4,
          "chartId": "speed-anatomy-first-text",
          "aN": 4,
          "bN": 4,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            2.84,
            6.38
          ],
          "bRange": [
            0.88,
            4.09
          ]
        },
        {
          "metric": "Output speed after the first text: visible tokens per second (calculation)",
          "aValue": 153.2,
          "bValue": 231.7,
          "unit": "tokens",
          "aDisplay": "153",
          "bDisplay": "232",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "llm-speed-anatomy",
          "n": 4,
          "chartId": "speed-anatomy-output-speed",
          "aN": 4,
          "bN": 4,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            152.6,
            216.1
          ],
          "bRange": [
            230.3,
            233
          ],
          "calculation": true
        },
        {
          "metric": "Output speed in characters per second after the first text (calculation)",
          "aValue": 547,
          "bValue": 517,
          "unit": "count",
          "aDisplay": "547",
          "bDisplay": "517",
          "winner": "unclear",
          "basis": "More or fewer count is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "llm-speed-anatomy",
          "chartId": "speed-anatomy-chars-per-second",
          "aN": 3,
          "bN": 4,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            546,
            548
          ],
          "bRange": [
            513,
            519
          ],
          "calculation": true
        },
        {
          "metric": "Time to first text as the prompt grows: 1k",
          "aValue": 1.93,
          "bValue": 1.45,
          "unit": "seconds",
          "aDisplay": "1.93 s",
          "bDisplay": "1.45 s",
          "winner": "unclear",
          "basis": "Only 3 runs per side; the run ranges (fastest to slowest) do not overlap (Claude Haiku 4.5 1.85 s to 2.04 s; Claude Sonnet 5.5 1.23 s to 1.72 s), but 3 runs cannot show a reliable difference. A range is not a confidence interval.",
          "studySlug": "llm-speed-anatomy",
          "n": 3,
          "chartId": "speed-anatomy-prompt-size",
          "aN": 3,
          "bN": 3,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            1.85,
            2.04
          ],
          "bRange": [
            1.23,
            1.72
          ],
          "calculation": true
        },
        {
          "metric": "Time to first text as the prompt grows: 16k",
          "aValue": 2.27,
          "bValue": 1.78,
          "unit": "seconds",
          "aDisplay": "2.27 s",
          "bDisplay": "1.78 s",
          "winner": "unclear",
          "basis": "Only 3 runs per side; the run ranges (fastest to slowest) do not overlap (Claude Haiku 4.5 2.22 s to 2.47 s; Claude Sonnet 5.5 1.64 s to 2.11 s), but 3 runs cannot show a reliable difference. A range is not a confidence interval.",
          "studySlug": "llm-speed-anatomy",
          "n": 3,
          "chartId": "speed-anatomy-prompt-size",
          "aN": 3,
          "bN": 3,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            2.22,
            2.47
          ],
          "bRange": [
            1.64,
            2.11
          ],
          "calculation": true
        },
        {
          "metric": "Time to first text as the prompt grows: 64k",
          "aValue": 2.78,
          "bValue": 3.07,
          "unit": "seconds",
          "aDisplay": "2.78 s",
          "bDisplay": "3.07 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Haiku 4.5 2.45 s to 2.89 s; Claude Sonnet 5.5 1.38 s to 3.61 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "llm-speed-anatomy",
          "n": 3,
          "chartId": "speed-anatomy-prompt-size",
          "aN": 3,
          "bN": 3,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            2.45,
            2.89
          ],
          "bRange": [
            1.38,
            3.61
          ],
          "calculation": true
        },
        {
          "metric": "Total time per call by prompt size (1k prompt)",
          "aValue": 2.34,
          "bValue": 1.78,
          "unit": "seconds",
          "aDisplay": "2.34 s",
          "bDisplay": "1.78 s",
          "winner": "unclear",
          "basis": "Only 3 runs per side; the run ranges (fastest to slowest) do not overlap (Claude Haiku 4.5 2.22 s to 2.46 s; Claude Sonnet 5.5 1.57 s to 2.12 s), but 3 runs cannot show a reliable difference. A range is not a confidence interval.",
          "studySlug": "llm-speed-anatomy",
          "n": 3,
          "chartId": "speed-anatomy-total-by-size",
          "aN": 3,
          "bN": 3,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            2.22,
            2.46
          ],
          "bRange": [
            1.57,
            2.12
          ]
        },
        {
          "metric": "Total time per call by prompt size (16k prompt)",
          "aValue": 2.79,
          "bValue": 2.1,
          "unit": "seconds",
          "aDisplay": "2.79 s",
          "bDisplay": "2.10 s",
          "winner": "unclear",
          "basis": "Only 3 runs per side; the run ranges (fastest to slowest) do not overlap (Claude Haiku 4.5 2.58 s to 2.84 s; Claude Sonnet 5.5 1.98 s to 2.48 s), but 3 runs cannot show a reliable difference. A range is not a confidence interval.",
          "studySlug": "llm-speed-anatomy",
          "n": 3,
          "chartId": "speed-anatomy-total-by-size",
          "aN": 3,
          "bN": 3,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            2.58,
            2.84
          ],
          "bRange": [
            1.98,
            2.48
          ]
        },
        {
          "metric": "Total time per call by prompt size (64k prompt)",
          "aValue": 3.13,
          "bValue": 3.44,
          "unit": "seconds",
          "aDisplay": "3.13 s",
          "bDisplay": "3.44 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Haiku 4.5 2.84 s to 3.28 s; Claude Sonnet 5.5 1.74 s to 4.38 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "llm-speed-anatomy",
          "n": 3,
          "chartId": "speed-anatomy-total-by-size",
          "aN": 3,
          "bN": 3,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            2.84,
            3.28
          ],
          "bRange": [
            1.74,
            4.38
          ]
        },
        {
          "metric": "Exact lookup answers at the 1k, 16k and 64k prompt-size targets",
          "aValue": 1,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "100% (9/9)",
          "bDisplay": "100% (9/9)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Haiku 4.5 70% to 100%; Claude Sonnet 5.5 70% to 100%), so this sample cannot separate them.",
          "studySlug": "llm-speed-anatomy",
          "n": 9,
          "chartId": "speed-anatomy-lookup-correct",
          "aN": 9,
          "bN": 9,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.7009,
            1
          ],
          "bRange": [
            0.7009,
            1
          ]
        },
        {
          "metric": "List-price cost of one call, Haiku 4.5 vs Sonnet 5.5, on each hard task (calculation): Interval merge fix",
          "aValue": 0.01789,
          "bValue": 0.00557,
          "unit": "usd",
          "aDisplay": "$0.018",
          "bDisplay": "$0.0056",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.018 vs $0.0056, 3.2x) is not tested against run-to-run variation.",
          "studySlug": "haiku-retry-or-escalate",
          "n": 3,
          "chartId": "retry-escalate-call-cost-by-task",
          "aN": 3,
          "bN": 3,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "calculation": true
        },
        {
          "metric": "List-price cost of one call, Haiku 4.5 vs Sonnet 5.5, on each hard task (calculation): DST day length",
          "aValue": 0.0293,
          "bValue": 0.02532,
          "unit": "usd",
          "aDisplay": "$0.029",
          "bDisplay": "$0.025",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.029 vs $0.025) is not tested against run-to-run variation.",
          "studySlug": "haiku-retry-or-escalate",
          "n": 3,
          "chartId": "retry-escalate-call-cost-by-task",
          "aN": 3,
          "bN": 3,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "calculation": true
        },
        {
          "metric": "List-price cost of one call, Haiku 4.5 vs Sonnet 5.5, on each hard task (calculation): CSV parser",
          "aValue": 0.02919,
          "bValue": 0.01464,
          "unit": "usd",
          "aDisplay": "$0.029",
          "bDisplay": "$0.015",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.029 vs $0.015) is not tested against run-to-run variation.",
          "studySlug": "haiku-retry-or-escalate",
          "n": 3,
          "chartId": "retry-escalate-call-cost-by-task",
          "aN": 3,
          "bN": 3,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "calculation": true
        },
        {
          "metric": "List-price cost of one call, Haiku 4.5 vs Sonnet 5.5, on each hard task (calculation): Event-loop order",
          "aValue": 0.03708,
          "bValue": 0.01588,
          "unit": "usd",
          "aDisplay": "$0.037",
          "bDisplay": "$0.016",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.037 vs $0.016, 2.3x) is not tested against run-to-run variation.",
          "studySlug": "haiku-retry-or-escalate",
          "n": 3,
          "chartId": "retry-escalate-call-cost-by-task",
          "aN": 3,
          "bN": 3,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "calculation": true
        },
        {
          "metric": "List-price cost of one call, Haiku 4.5 vs Sonnet 5.5, on each hard task (calculation): Room schedule",
          "aValue": 0.03578,
          "bValue": 0.01243,
          "unit": "usd",
          "aDisplay": "$0.036",
          "bDisplay": "$0.012",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.036 vs $0.012, 2.9x) is not tested against run-to-run variation.",
          "studySlug": "haiku-retry-or-escalate",
          "n": 3,
          "chartId": "retry-escalate-call-cost-by-task",
          "aN": 3,
          "bN": 3,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "calculation": true
        },
        {
          "metric": "List-price cost of one call, Haiku 4.5 vs Sonnet 5.5, on each hard task (calculation): SemVer regex",
          "aValue": 0.04217,
          "bValue": 0.00514,
          "unit": "usd",
          "aDisplay": "$0.042",
          "bDisplay": "$0.0051",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.042 vs $0.0051, 8.2x) is not tested against run-to-run variation.",
          "studySlug": "haiku-retry-or-escalate",
          "n": 3,
          "chartId": "retry-escalate-call-cost-by-task",
          "aN": 3,
          "bN": 3,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "calculation": true
        },
        {
          "metric": "List-price cost of one call, Haiku 4.5 vs Sonnet 5.5, on each hard task (calculation): Money refactor",
          "aValue": 0.02039,
          "bValue": 0.00961,
          "unit": "usd",
          "aDisplay": "$0.020",
          "bDisplay": "$0.0096",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.020 vs $0.0096, 2.1x) is not tested against run-to-run variation.",
          "studySlug": "haiku-retry-or-escalate",
          "n": 3,
          "chartId": "retry-escalate-call-cost-by-task",
          "aN": 3,
          "bN": 3,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "calculation": true
        },
        {
          "metric": "List-price cost of one call, Haiku 4.5 vs Sonnet 5.5, on each hard task (calculation): SQLite report query",
          "aValue": 0.03648,
          "bValue": 0.01719,
          "unit": "usd",
          "aDisplay": "$0.036",
          "bDisplay": "$0.017",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.036 vs $0.017, 2.1x) is not tested against run-to-run variation.",
          "studySlug": "haiku-retry-or-escalate",
          "n": 3,
          "chartId": "retry-escalate-call-cost-by-task",
          "aN": 3,
          "bN": 3,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "calculation": true
        },
        {
          "metric": "Pass rate on 4 harder tasks (Strict pass)",
          "aValue": 0,
          "bValue": 0.375,
          "unit": "rate",
          "aDisplay": "0% (0/12)",
          "bDisplay": "38% (6/16)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Haiku 4.5 0% to 24%; Claude Sonnet 5.5 18% to 61%), so this sample cannot separate them.",
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-pass-rate",
          "aN": 12,
          "bN": 16,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0,
            0.2425
          ],
          "bRange": [
            0.1848,
            0.6136
          ]
        },
        {
          "metric": "Pass rate on 4 harder tasks (Lenient (format misses counted))",
          "aValue": 0,
          "bValue": 0.375,
          "unit": "rate",
          "aDisplay": "0% (0/12)",
          "bDisplay": "38% (6/16)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Haiku 4.5 0% to 24%; Claude Sonnet 5.5 18% to 61%), so this sample cannot separate them.",
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-pass-rate",
          "aN": 12,
          "bN": 16,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0,
            0.2425
          ],
          "bRange": [
            0.1848,
            0.6136
          ]
        },
        {
          "metric": "Calls that tried a tool although tools were off",
          "aValue": 0.0833,
          "bValue": 0.3125,
          "unit": "rate",
          "aDisplay": "8% (1/12)",
          "bDisplay": "31% (5/16)",
          "winner": "unclear",
          "basis": "More or fewer rate is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-tool-attempts",
          "aN": 12,
          "bN": 16,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.0149,
            0.3539
          ],
          "bRange": [
            0.1416,
            0.556
          ]
        },
        {
          "metric": "Strict pass rate by task: 10x10 nonogram",
          "aValue": 0,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "0% (0/3)",
          "bDisplay": "100% (4/4)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Haiku 4.5 0% to 56%; Claude Sonnet 5.5 51% to 100%), so this sample cannot separate them.",
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-pass-by-task",
          "aN": 3,
          "bN": 4,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0,
            0.5615
          ],
          "bRange": [
            0.5101,
            1
          ]
        },
        {
          "metric": "Strict pass rate by task: Sudoku, 22 givens",
          "aValue": 0,
          "bValue": 0,
          "unit": "rate",
          "aDisplay": "0% (0/3)",
          "bDisplay": "0% (0/4)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Haiku 4.5 0% to 56%; Claude Sonnet 5.5 0% to 49%), so this sample cannot separate them.",
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-pass-by-task",
          "aN": 3,
          "bN": 4,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0,
            0.5615
          ],
          "bRange": [
            0,
            0.4899
          ]
        },
        {
          "metric": "Strict pass rate by task: 6x6 Skyscrapers",
          "aValue": 0,
          "bValue": 0,
          "unit": "rate",
          "aDisplay": "0% (0/3)",
          "bDisplay": "0% (0/4)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Haiku 4.5 0% to 56%; Claude Sonnet 5.5 0% to 49%), so this sample cannot separate them.",
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-pass-by-task",
          "aN": 3,
          "bN": 4,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0,
            0.5615
          ],
          "bRange": [
            0,
            0.4899
          ]
        },
        {
          "metric": "Strict pass rate by task: Seeded shuffle output",
          "aValue": 0,
          "bValue": 0.5,
          "unit": "rate",
          "aDisplay": "0% (0/3)",
          "bDisplay": "50% (2/4)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Haiku 4.5 0% to 56%; Claude Sonnet 5.5 15% to 85%), so this sample cannot separate them.",
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-pass-by-task",
          "aN": 3,
          "bN": 4,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0,
            0.5615
          ],
          "bRange": [
            0.15,
            0.85
          ]
        },
        {
          "metric": "Total time per call on harder tasks",
          "aValue": 108.98,
          "bValue": 70.43,
          "unit": "seconds",
          "aDisplay": "109.0 s",
          "bDisplay": "70.4 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Haiku 4.5 25.7 s to 223.9 s; Claude Sonnet 5.5 4.32 s to 210.1 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-total-latency",
          "aN": 10,
          "bN": 12,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            25.73,
            223.95
          ],
          "bRange": [
            4.32,
            210.08
          ]
        },
        {
          "metric": "Output tokens per call on harder tasks (Output tokens)",
          "aValue": 12508,
          "bValue": 9287,
          "unit": "tokens",
          "aDisplay": "12,508",
          "bDisplay": "9,287",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-output-tokens",
          "aN": 10,
          "bN": 12,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            2965,
            26532
          ],
          "bRange": [
            407,
            27921
          ]
        }
      ]
    },
    {
      "slug": "claude-opus-5-5-vs-claude-fable-5-1",
      "a": "claude-opus-5-5",
      "b": "claude-fable-5-1",
      "title": "Claude Opus 5.5 vs Claude Fable 5.1",
      "seoTitle": "Claude Opus 5.5 vs Claude Fable 5.1: measured benchmarks",
      "description": "Claude Opus 5.5 vs Claude Fable 5.1: 12 measured metrics from 3 studies (Pass rate on five validated tasks; more), with sample sizes and intervals.",
      "verdict": "Claude Opus 5.5 and Claude Fable 5.1 share 12 measured metrics and 11 list-price calculations from 4 studies. No row separates them: every interval or run range overlaps, too few runs were recorded, no interval was recorded, or more is not better for that metric. The rows are 5 ties and 18 unclear; each row says why. Calculation rows are derived from list prices and recorded counts; they are not bills or runs. Some rows rest on small samples (n = 4 at the smallest).",
      "rows": [
        {
          "metric": "Pass rate on five validated tasks",
          "aValue": 1,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "100% (15/15)",
          "bDisplay": "100% (15/15)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Opus 5.5 80% to 100%; Claude Fable 5.1 80% to 100%), so this sample cannot separate them.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-pass-rate",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code · five short validated tasks",
          "bContext": "Claude Code · five short validated tasks",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.7961,
            1
          ],
          "bRange": [
            0.7961,
            1
          ]
        },
        {
          "metric": "Total time per call",
          "aValue": 2.75,
          "bValue": 1.94,
          "unit": "seconds",
          "aDisplay": "2.75 s",
          "bDisplay": "1.94 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Opus 5.5 2.47 s to 8.91 s; Claude Fable 5.1 1.41 s to 9.83 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-total-latency",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code · five short validated tasks",
          "bContext": "Claude Code · five short validated tasks",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            2.47,
            8.91
          ],
          "bRange": [
            1.41,
            9.83
          ]
        },
        {
          "metric": "Time to first useful output",
          "aValue": 1.92,
          "bValue": 1.2,
          "unit": "seconds",
          "aDisplay": "1.92 s",
          "bDisplay": "1.20 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Opus 5.5 1.56 s to 7.23 s; Claude Fable 5.1 0.95 s to 7.90 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-first-useful-latency",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code · five short validated tasks",
          "bContext": "Claude Code · five short validated tasks",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            1.56,
            7.23
          ],
          "bRange": [
            0.95,
            7.9
          ]
        },
        {
          "metric": "Input tokens per call: what the CLI sends (Cache read)",
          "aValue": 1401,
          "bValue": 2760,
          "unit": "tokens",
          "aDisplay": "1,401",
          "bDisplay": "2,760",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-input-tokens",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code · five short validated tasks",
          "bContext": "Claude Code · five short validated tasks"
        },
        {
          "metric": "Input tokens per call: what the CLI sends (Other input)",
          "aValue": 680,
          "bValue": 473,
          "unit": "tokens",
          "aDisplay": "680",
          "bDisplay": "473",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-input-tokens",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code · five short validated tasks",
          "bContext": "Claude Code · five short validated tasks"
        },
        {
          "metric": "Output tokens per call (Output tokens)",
          "aValue": 64,
          "bValue": 64,
          "unit": "tokens",
          "aDisplay": "64",
          "bDisplay": "64",
          "winner": "tie",
          "basis": "Same value. More or fewer is not better by itself for this metric.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-output-tokens",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code · five short validated tasks",
          "bContext": "Claude Code · five short validated tasks"
        },
        {
          "metric": "List-price cost per call (calculation)",
          "aValue": 0.00688,
          "bValue": 0.00987,
          "unit": "usd",
          "aDisplay": "$0.0069",
          "bDisplay": "$0.0099",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Opus 5.5 $0.0059 to $0.022; Claude Fable 5.1 $0.0049 to $0.058); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-list-price-per-call",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code · five short validated tasks",
          "bContext": "Claude Code · five short validated tasks",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            0.00592,
            0.02226
          ],
          "bRange": [
            0.0049,
            0.05843
          ],
          "calculation": true
        },
        {
          "metric": "List-price cost per passing answer (calculation)",
          "aValue": 0.01009,
          "bValue": 0.02054,
          "unit": "usd",
          "aDisplay": "$0.010",
          "bDisplay": "$0.021",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.010 vs $0.021, 2.0x) is not tested against run-to-run variation.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-cost-per-pass",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code · five short validated tasks",
          "bContext": "Claude Code · five short validated tasks",
          "calculation": true
        },
        {
          "metric": "Pass rate on eight hard tasks (Strict pass)",
          "aValue": 1,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "100% (24/24)",
          "bDisplay": "100% (24/24)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Opus 5.5 86% to 100%; Claude Fable 5.1 86% to 100%), so this sample cannot separate them.",
          "studySlug": "hard-model-head-to-head",
          "n": 24,
          "chartId": "hard-h2h-pass-rate",
          "aN": 24,
          "bN": 24,
          "aContext": "Claude Code · eight hard validated tasks",
          "bContext": "Claude Code · eight hard validated tasks",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.862,
            1
          ],
          "bRange": [
            0.862,
            1
          ]
        },
        {
          "metric": "Pass rate on eight hard tasks (Lenient (format misses counted))",
          "aValue": 1,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "100% (24/24)",
          "bDisplay": "100% (24/24)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Opus 5.5 86% to 100%; Claude Fable 5.1 86% to 100%), so this sample cannot separate them.",
          "studySlug": "hard-model-head-to-head",
          "n": 24,
          "chartId": "hard-h2h-pass-rate",
          "aN": 24,
          "bN": 24,
          "aContext": "Claude Code · eight hard validated tasks",
          "bContext": "Claude Code · eight hard validated tasks",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.862,
            1
          ],
          "bRange": [
            0.862,
            1
          ]
        },
        {
          "metric": "Total time per call on hard tasks (separate batches)",
          "aValue": 9.18,
          "bValue": 16.13,
          "unit": "seconds",
          "aDisplay": "9.18 s",
          "bDisplay": "16.1 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Opus 5.5 4.24 s to 27.2 s; Claude Fable 5.1 4.46 s to 90.0 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "hard-model-head-to-head",
          "n": 24,
          "chartId": "hard-h2h-total-latency",
          "aN": 24,
          "bN": 24,
          "aContext": "Claude Code · eight hard validated tasks",
          "bContext": "Claude Code · eight hard validated tasks",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            4.24,
            27.21
          ],
          "bRange": [
            4.46,
            90
          ]
        },
        {
          "metric": "Time to first useful output on hard tasks",
          "aValue": 6.78,
          "bValue": 11.63,
          "unit": "seconds",
          "aDisplay": "6.78 s",
          "bDisplay": "11.6 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Opus 5.5 2.39 s to 21.8 s; Claude Fable 5.1 2.00 s to 85.3 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "hard-model-head-to-head",
          "n": 24,
          "chartId": "hard-h2h-first-useful-latency",
          "aN": 24,
          "bN": 24,
          "aContext": "Claude Code · eight hard validated tasks",
          "bContext": "Claude Code · eight hard validated tasks",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            2.39,
            21.77
          ],
          "bRange": [
            2,
            85.33
          ]
        },
        {
          "metric": "Output tokens per call on hard tasks (Output tokens)",
          "aValue": 945,
          "bValue": 1366,
          "unit": "tokens",
          "aDisplay": "945",
          "bDisplay": "1,366",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "hard-model-head-to-head",
          "n": 24,
          "chartId": "hard-h2h-output-tokens",
          "aN": 24,
          "bN": 24,
          "aContext": "Claude Code · eight hard validated tasks",
          "bContext": "Claude Code · eight hard validated tasks"
        },
        {
          "metric": "List-price cost per strict pass on hard tasks (calculation)",
          "aValue": 0.02824,
          "bValue": 0.09331,
          "unit": "usd",
          "aDisplay": "$0.028",
          "bDisplay": "$0.093",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.028 vs $0.093, 3.3x) is not tested against run-to-run variation.",
          "studySlug": "hard-model-head-to-head",
          "n": 24,
          "chartId": "hard-h2h-cost-per-pass",
          "aN": 24,
          "bN": 24,
          "aContext": "Claude Code · eight hard validated tasks",
          "bContext": "Claude Code · eight hard validated tasks",
          "calculation": true
        },
        {
          "metric": "Reasoning share of output tokens per call on hard tasks (calculation)",
          "aValue": 54.79,
          "bValue": 64.24,
          "unit": "percent",
          "aDisplay": "54.8%",
          "bDisplay": "64.2%",
          "winner": "unclear",
          "basis": "More or fewer percent is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "n": 24,
          "chartId": "thinking-bill-share",
          "aN": 24,
          "bN": 24,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            29.92,
            95.6
          ],
          "bRange": [
            23.44,
            97.19
          ],
          "calculation": true
        },
        {
          "metric": "List-price cost per call: reasoning, remaining output and input (calculation) (Reasoning (output tokens))",
          "aValue": 0.012528,
          "bValue": 0.053696,
          "unit": "usd",
          "aDisplay": "$0.013",
          "bDisplay": "$0.054",
          "winner": "unclear",
          "basis": "More or fewer usd is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "n": 24,
          "chartId": "thinking-bill-cost-per-call",
          "aN": 24,
          "bN": 24,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "calculation": true
        },
        {
          "metric": "List-price cost per call: reasoning, remaining output and input (calculation) (Remaining output (visible-answer estimate))",
          "aValue": 0.008003,
          "bValue": 0.018702,
          "unit": "usd",
          "aDisplay": "$0.0080",
          "bDisplay": "$0.019",
          "winner": "unclear",
          "basis": "More or fewer usd is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "n": 24,
          "chartId": "thinking-bill-cost-per-call",
          "aN": 24,
          "bN": 24,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "calculation": true
        },
        {
          "metric": "List-price cost per call: reasoning, remaining output and input (calculation) (Input (prompt, cache priced))",
          "aValue": 0.00771,
          "bValue": 0.02091,
          "unit": "usd",
          "aDisplay": "$0.0077",
          "bDisplay": "$0.021",
          "winner": "unclear",
          "basis": "More or fewer usd is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "n": 24,
          "chartId": "thinking-bill-cost-per-call",
          "aN": 24,
          "bN": 24,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "calculation": true
        },
        {
          "metric": "Reasoning share on short tasks vs hard tasks (calculation) (Eight hard tasks)",
          "aValue": 54.79,
          "bValue": 64.24,
          "unit": "percent",
          "aDisplay": "54.8%",
          "bDisplay": "64.2%",
          "winner": "unclear",
          "basis": "More or fewer percent is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "n": 24,
          "chartId": "thinking-bill-short-vs-hard",
          "aN": 24,
          "bN": 24,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            29.92,
            95.6
          ],
          "bRange": [
            23.44,
            97.19
          ],
          "calculation": true
        },
        {
          "metric": "Reasoning share on short tasks vs hard tasks (calculation) (Five short tasks)",
          "aValue": 0,
          "bValue": 0,
          "unit": "percent",
          "aDisplay": "0%",
          "bDisplay": "0%",
          "winner": "tie",
          "basis": "Same value. More or fewer is not better by itself for this metric.",
          "studySlug": "thinking-token-bill",
          "n": 15,
          "chartId": "thinking-bill-short-vs-hard",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            0,
            93.33
          ],
          "bRange": [
            0,
            74.01
          ],
          "calculation": true
        },
        {
          "metric": "Time to first text: a 250-line answer, six models",
          "aValue": 1.97,
          "bValue": 4.43,
          "unit": "seconds",
          "aDisplay": "1.97 s",
          "bDisplay": "4.43 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Opus 5.5 1.70 s to 2.35 s; Claude Fable 5.1 2.27 s to 4.64 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "llm-speed-anatomy",
          "n": 4,
          "chartId": "speed-anatomy-first-text",
          "aN": 4,
          "bN": 4,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            1.7,
            2.35
          ],
          "bRange": [
            2.27,
            4.64
          ]
        },
        {
          "metric": "Output speed after the first text: visible tokens per second (calculation)",
          "aValue": 155.5,
          "bValue": 122.6,
          "unit": "tokens",
          "aDisplay": "156",
          "bDisplay": "123",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "llm-speed-anatomy",
          "n": 4,
          "chartId": "speed-anatomy-output-speed",
          "aN": 4,
          "bN": 4,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            154.6,
            156.4
          ],
          "bRange": [
            120.9,
            131.4
          ],
          "calculation": true
        },
        {
          "metric": "Output speed in characters per second after the first text (calculation)",
          "aValue": 347,
          "bValue": 273,
          "unit": "count",
          "aDisplay": "347",
          "bDisplay": "273",
          "winner": "unclear",
          "basis": "More or fewer count is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "llm-speed-anatomy",
          "n": 4,
          "chartId": "speed-anatomy-chars-per-second",
          "aN": 4,
          "bN": 4,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            345,
            349
          ],
          "bRange": [
            270,
            293
          ],
          "calculation": true
        }
      ]
    },
    {
      "slug": "claude-sonnet-5-5-vs-claude-fable-5-1",
      "a": "claude-sonnet-5-5",
      "b": "claude-fable-5-1",
      "title": "Claude Sonnet 5.5 vs Claude Fable 5.1",
      "seoTitle": "Claude Sonnet 5.5 vs Claude Fable 5.1: measured benchmarks",
      "description": "Claude Sonnet 5.5 vs Claude Fable 5.1: 12 measured metrics from 3 studies (Pass rate on five validated tasks; more), with sample sizes and intervals.",
      "verdict": "Claude Sonnet 5.5 and Claude Fable 5.1 share 12 measured metrics and 11 list-price calculations from 4 studies. No row separates them: every interval or run range overlaps, too few runs were recorded, no interval was recorded, or more is not better for that metric. The rows are 4 ties and 19 unclear; each row says why. Calculation rows are derived from list prices and recorded counts; they are not bills or runs. Some rows rest on small samples (n = 4 at the smallest).",
      "rows": [
        {
          "metric": "Pass rate on five validated tasks",
          "aValue": 0.8,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "80% (12/15)",
          "bDisplay": "100% (15/15)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Sonnet 5.5 55% to 93%; Claude Fable 5.1 80% to 100%), so this sample cannot separate them.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-pass-rate",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code · five short validated tasks",
          "bContext": "Claude Code · five short validated tasks",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.5481,
            0.9295
          ],
          "bRange": [
            0.7961,
            1
          ]
        },
        {
          "metric": "Total time per call",
          "aValue": 2.31,
          "bValue": 1.94,
          "unit": "seconds",
          "aDisplay": "2.31 s",
          "bDisplay": "1.94 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Sonnet 5.5 2.17 s to 7.73 s; Claude Fable 5.1 1.41 s to 9.83 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-total-latency",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code · five short validated tasks",
          "bContext": "Claude Code · five short validated tasks",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            2.17,
            7.73
          ],
          "bRange": [
            1.41,
            9.83
          ]
        },
        {
          "metric": "Time to first useful output",
          "aValue": 1.56,
          "bValue": 1.2,
          "unit": "seconds",
          "aDisplay": "1.56 s",
          "bDisplay": "1.20 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Sonnet 5.5 0.99 s to 6.39 s; Claude Fable 5.1 0.95 s to 7.90 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-first-useful-latency",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code · five short validated tasks",
          "bContext": "Claude Code · five short validated tasks",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            0.99,
            6.39
          ],
          "bRange": [
            0.95,
            7.9
          ]
        },
        {
          "metric": "Input tokens per call: what the CLI sends (Cache read)",
          "aValue": 1401,
          "bValue": 2760,
          "unit": "tokens",
          "aDisplay": "1,401",
          "bDisplay": "2,760",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-input-tokens",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code · five short validated tasks",
          "bContext": "Claude Code · five short validated tasks"
        },
        {
          "metric": "Input tokens per call: what the CLI sends (Other input)",
          "aValue": 685,
          "bValue": 473,
          "unit": "tokens",
          "aDisplay": "685",
          "bDisplay": "473",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-input-tokens",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code · five short validated tasks",
          "bContext": "Claude Code · five short validated tasks"
        },
        {
          "metric": "Output tokens per call (Output tokens)",
          "aValue": 107,
          "bValue": 64,
          "unit": "tokens",
          "aDisplay": "107",
          "bDisplay": "64",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-output-tokens",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code · five short validated tasks",
          "bContext": "Claude Code · five short validated tasks"
        },
        {
          "metric": "List-price cost per call (calculation)",
          "aValue": 0.0036,
          "bValue": 0.00987,
          "unit": "usd",
          "aDisplay": "$0.0036",
          "bDisplay": "$0.0099",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Sonnet 5.5 $0.0034 to $0.010; Claude Fable 5.1 $0.0049 to $0.058); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-list-price-per-call",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code · five short validated tasks",
          "bContext": "Claude Code · five short validated tasks",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            0.00342,
            0.01021
          ],
          "bRange": [
            0.0049,
            0.05843
          ],
          "calculation": true
        },
        {
          "metric": "List-price cost per passing answer (calculation)",
          "aValue": 0.00624,
          "bValue": 0.02054,
          "unit": "usd",
          "aDisplay": "$0.0062",
          "bDisplay": "$0.021",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.0062 vs $0.021, 3.3x) is not tested against run-to-run variation.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-cost-per-pass",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code · five short validated tasks",
          "bContext": "Claude Code · five short validated tasks",
          "calculation": true
        },
        {
          "metric": "Pass rate on eight hard tasks (Strict pass)",
          "aValue": 1,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "100% (24/24)",
          "bDisplay": "100% (24/24)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Sonnet 5.5 86% to 100%; Claude Fable 5.1 86% to 100%), so this sample cannot separate them.",
          "studySlug": "hard-model-head-to-head",
          "n": 24,
          "chartId": "hard-h2h-pass-rate",
          "aN": 24,
          "bN": 24,
          "aContext": "Claude Code · eight hard validated tasks",
          "bContext": "Claude Code · eight hard validated tasks",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.862,
            1
          ],
          "bRange": [
            0.862,
            1
          ]
        },
        {
          "metric": "Pass rate on eight hard tasks (Lenient (format misses counted))",
          "aValue": 1,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "100% (24/24)",
          "bDisplay": "100% (24/24)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Sonnet 5.5 86% to 100%; Claude Fable 5.1 86% to 100%), so this sample cannot separate them.",
          "studySlug": "hard-model-head-to-head",
          "n": 24,
          "chartId": "hard-h2h-pass-rate",
          "aN": 24,
          "bN": 24,
          "aContext": "Claude Code · eight hard validated tasks",
          "bContext": "Claude Code · eight hard validated tasks",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.862,
            1
          ],
          "bRange": [
            0.862,
            1
          ]
        },
        {
          "metric": "Total time per call on hard tasks (separate batches)",
          "aValue": 7.75,
          "bValue": 16.13,
          "unit": "seconds",
          "aDisplay": "7.75 s",
          "bDisplay": "16.1 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Sonnet 5.5 2.26 s to 34.8 s; Claude Fable 5.1 4.46 s to 90.0 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "hard-model-head-to-head",
          "n": 24,
          "chartId": "hard-h2h-total-latency",
          "aN": 24,
          "bN": 24,
          "aContext": "Claude Code · eight hard validated tasks",
          "bContext": "Claude Code · eight hard validated tasks",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            2.26,
            34.79
          ],
          "bRange": [
            4.46,
            90
          ]
        },
        {
          "metric": "Time to first useful output on hard tasks",
          "aValue": 5.95,
          "bValue": 11.63,
          "unit": "seconds",
          "aDisplay": "5.95 s",
          "bDisplay": "11.6 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Sonnet 5.5 0.86 s to 30.6 s; Claude Fable 5.1 2.00 s to 85.3 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "hard-model-head-to-head",
          "n": 24,
          "chartId": "hard-h2h-first-useful-latency",
          "aN": 24,
          "bN": 24,
          "aContext": "Claude Code · eight hard validated tasks",
          "bContext": "Claude Code · eight hard validated tasks",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            0.86,
            30.57
          ],
          "bRange": [
            2,
            85.33
          ]
        },
        {
          "metric": "Output tokens per call on hard tasks (Output tokens)",
          "aValue": 1050,
          "bValue": 1366,
          "unit": "tokens",
          "aDisplay": "1,050",
          "bDisplay": "1,366",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "hard-model-head-to-head",
          "n": 24,
          "chartId": "hard-h2h-output-tokens",
          "aN": 24,
          "bN": 24,
          "aContext": "Claude Code · eight hard validated tasks",
          "bContext": "Claude Code · eight hard validated tasks"
        },
        {
          "metric": "List-price cost per strict pass on hard tasks (calculation)",
          "aValue": 0.01435,
          "bValue": 0.09331,
          "unit": "usd",
          "aDisplay": "$0.014",
          "bDisplay": "$0.093",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.014 vs $0.093, 6.5x) is not tested against run-to-run variation.",
          "studySlug": "hard-model-head-to-head",
          "n": 24,
          "chartId": "hard-h2h-cost-per-pass",
          "aN": 24,
          "bN": 24,
          "aContext": "Claude Code · eight hard validated tasks",
          "bContext": "Claude Code · eight hard validated tasks",
          "calculation": true
        },
        {
          "metric": "Reasoning share of output tokens per call on hard tasks (calculation)",
          "aValue": 54.54,
          "bValue": 64.24,
          "unit": "percent",
          "aDisplay": "54.5%",
          "bDisplay": "64.2%",
          "winner": "unclear",
          "basis": "More or fewer percent is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "n": 24,
          "chartId": "thinking-bill-share",
          "aN": 24,
          "bN": 24,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            0,
            95.91
          ],
          "bRange": [
            23.44,
            97.19
          ],
          "calculation": true
        },
        {
          "metric": "List-price cost per call: reasoning, remaining output and input (calculation) (Reasoning (output tokens))",
          "aValue": 0.006665,
          "bValue": 0.053696,
          "unit": "usd",
          "aDisplay": "$0.0067",
          "bDisplay": "$0.054",
          "winner": "unclear",
          "basis": "More or fewer usd is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "n": 24,
          "chartId": "thinking-bill-cost-per-call",
          "aN": 24,
          "bN": 24,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "calculation": true
        },
        {
          "metric": "List-price cost per call: reasoning, remaining output and input (calculation) (Remaining output (visible-answer estimate))",
          "aValue": 0.003672,
          "bValue": 0.018702,
          "unit": "usd",
          "aDisplay": "$0.0037",
          "bDisplay": "$0.019",
          "winner": "unclear",
          "basis": "More or fewer usd is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "n": 24,
          "chartId": "thinking-bill-cost-per-call",
          "aN": 24,
          "bN": 24,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "calculation": true
        },
        {
          "metric": "List-price cost per call: reasoning, remaining output and input (calculation) (Input (prompt, cache priced))",
          "aValue": 0.004012,
          "bValue": 0.02091,
          "unit": "usd",
          "aDisplay": "$0.0040",
          "bDisplay": "$0.021",
          "winner": "unclear",
          "basis": "More or fewer usd is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "n": 24,
          "chartId": "thinking-bill-cost-per-call",
          "aN": 24,
          "bN": 24,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "calculation": true
        },
        {
          "metric": "Reasoning share on short tasks vs hard tasks (calculation) (Eight hard tasks)",
          "aValue": 54.54,
          "bValue": 64.24,
          "unit": "percent",
          "aDisplay": "54.5%",
          "bDisplay": "64.2%",
          "winner": "unclear",
          "basis": "More or fewer percent is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "n": 24,
          "chartId": "thinking-bill-short-vs-hard",
          "aN": 24,
          "bN": 24,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            0,
            95.91
          ],
          "bRange": [
            23.44,
            97.19
          ],
          "calculation": true
        },
        {
          "metric": "Reasoning share on short tasks vs hard tasks (calculation) (Five short tasks)",
          "aValue": 0,
          "bValue": 0,
          "unit": "percent",
          "aDisplay": "0%",
          "bDisplay": "0%",
          "winner": "tie",
          "basis": "Same value. More or fewer is not better by itself for this metric.",
          "studySlug": "thinking-token-bill",
          "n": 15,
          "chartId": "thinking-bill-short-vs-hard",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            0,
            72.75
          ],
          "bRange": [
            0,
            74.01
          ],
          "calculation": true
        },
        {
          "metric": "Time to first text: a 250-line answer, six models",
          "aValue": 1.96,
          "bValue": 4.43,
          "unit": "seconds",
          "aDisplay": "1.96 s",
          "bDisplay": "4.43 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Sonnet 5.5 0.88 s to 4.09 s; Claude Fable 5.1 2.27 s to 4.64 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "llm-speed-anatomy",
          "n": 4,
          "chartId": "speed-anatomy-first-text",
          "aN": 4,
          "bN": 4,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            0.88,
            4.09
          ],
          "bRange": [
            2.27,
            4.64
          ]
        },
        {
          "metric": "Output speed after the first text: visible tokens per second (calculation)",
          "aValue": 231.7,
          "bValue": 122.6,
          "unit": "tokens",
          "aDisplay": "232",
          "bDisplay": "123",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "llm-speed-anatomy",
          "n": 4,
          "chartId": "speed-anatomy-output-speed",
          "aN": 4,
          "bN": 4,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            230.3,
            233
          ],
          "bRange": [
            120.9,
            131.4
          ],
          "calculation": true
        },
        {
          "metric": "Output speed in characters per second after the first text (calculation)",
          "aValue": 517,
          "bValue": 273,
          "unit": "count",
          "aDisplay": "517",
          "bDisplay": "273",
          "winner": "unclear",
          "basis": "More or fewer count is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "llm-speed-anatomy",
          "n": 4,
          "chartId": "speed-anatomy-chars-per-second",
          "aN": 4,
          "bN": 4,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            513,
            519
          ],
          "bRange": [
            270,
            293
          ],
          "calculation": true
        }
      ]
    },
    {
      "slug": "claude-haiku-4-5-vs-claude-opus-5-5",
      "a": "claude-haiku-4-5",
      "b": "claude-opus-5-5",
      "title": "Claude Haiku 4.5 vs Claude Opus 5.5",
      "seoTitle": "Claude Haiku 4.5 vs Claude Opus 5.5: measured benchmarks",
      "description": "Claude Haiku 4.5 vs Claude Opus 5.5: 25 measured metrics from 4 studies (Pass rate on five validated tasks; more), with sample sizes and intervals.",
      "verdict": "Claude Haiku 4.5 and Claude Opus 5.5 share 25 measured metrics and 14 list-price calculations from 5 studies. Claude Opus 5.5 leads on 3 rows: Pass rate on eight hard tasks (Strict pass), 100% (24/24) vs 46% (11/24); Pass rate on eight hard tasks (Lenient (format misses counted)), 100% (24/24) vs 67% (16/24); Pass rate on 4 harder tasks (Lenient (format misses counted)), 50% (6/12) vs 0% (0/12). On those rows the 95% intervals do not overlap. The other rows are 7 ties and 29 unclear; each row says why. Calculation rows are derived from list prices and recorded counts; they are not bills or runs. Some rows rest on small samples (n = 3 at the smallest).",
      "rows": [
        {
          "metric": "Pass rate on five validated tasks",
          "aValue": 1,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "100% (15/15)",
          "bDisplay": "100% (15/15)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Haiku 4.5 80% to 100%; Claude Opus 5.5 80% to 100%), so this sample cannot separate them.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-pass-rate",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code · five short validated tasks",
          "bContext": "Claude Code · five short validated tasks",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.7961,
            1
          ],
          "bRange": [
            0.7961,
            1
          ]
        },
        {
          "metric": "Total time per call",
          "aValue": 4.43,
          "bValue": 2.75,
          "unit": "seconds",
          "aDisplay": "4.43 s",
          "bDisplay": "2.75 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Haiku 4.5 3.16 s to 23.6 s; Claude Opus 5.5 2.47 s to 8.91 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-total-latency",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code · five short validated tasks",
          "bContext": "Claude Code · five short validated tasks",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            3.16,
            23.57
          ],
          "bRange": [
            2.47,
            8.91
          ]
        },
        {
          "metric": "Time to first useful output",
          "aValue": 3.63,
          "bValue": 1.92,
          "unit": "seconds",
          "aDisplay": "3.63 s",
          "bDisplay": "1.92 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Haiku 4.5 2.78 s to 22.3 s; Claude Opus 5.5 1.56 s to 7.23 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-first-useful-latency",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code · five short validated tasks",
          "bContext": "Claude Code · five short validated tasks",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            2.78,
            22.27
          ],
          "bRange": [
            1.56,
            7.23
          ]
        },
        {
          "metric": "Input tokens per call: what the CLI sends (Cache read)",
          "aValue": 0,
          "bValue": 1401,
          "unit": "tokens",
          "aDisplay": "0",
          "bDisplay": "1,401",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-input-tokens",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code · five short validated tasks",
          "bContext": "Claude Code · five short validated tasks"
        },
        {
          "metric": "Input tokens per call: what the CLI sends (Other input)",
          "aValue": 3790,
          "bValue": 680,
          "unit": "tokens",
          "aDisplay": "3,790",
          "bDisplay": "680",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-input-tokens",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code · five short validated tasks",
          "bContext": "Claude Code · five short validated tasks"
        },
        {
          "metric": "Output tokens per call (Output tokens)",
          "aValue": 367,
          "bValue": 64,
          "unit": "tokens",
          "aDisplay": "367",
          "bDisplay": "64",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-output-tokens",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code · five short validated tasks",
          "bContext": "Claude Code · five short validated tasks"
        },
        {
          "metric": "List-price cost per call (calculation)",
          "aValue": 0.00566,
          "bValue": 0.00688,
          "unit": "usd",
          "aDisplay": "$0.0057",
          "bDisplay": "$0.0069",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Haiku 4.5 $0.0051 to $0.018; Claude Opus 5.5 $0.0059 to $0.022); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-list-price-per-call",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code · five short validated tasks",
          "bContext": "Claude Code · five short validated tasks",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            0.00513,
            0.01804
          ],
          "bRange": [
            0.00592,
            0.02226
          ],
          "calculation": true
        },
        {
          "metric": "List-price cost per passing answer (calculation)",
          "aValue": 0.00836,
          "bValue": 0.01009,
          "unit": "usd",
          "aDisplay": "$0.0084",
          "bDisplay": "$0.010",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.0084 vs $0.010) is not tested against run-to-run variation.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-cost-per-pass",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code · five short validated tasks",
          "bContext": "Claude Code · five short validated tasks",
          "calculation": true
        },
        {
          "metric": "Pass rate on eight hard tasks (Strict pass)",
          "aValue": 0.4583,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "46% (11/24)",
          "bDisplay": "100% (24/24)",
          "winner": "b",
          "basis": "The 95% intervals do not overlap (Claude Haiku 4.5 28% to 65%; Claude Opus 5.5 86% to 100%).",
          "studySlug": "hard-model-head-to-head",
          "n": 24,
          "chartId": "hard-h2h-pass-rate",
          "aN": 24,
          "bN": 24,
          "aContext": "Claude Code · eight hard validated tasks",
          "bContext": "Claude Code · eight hard validated tasks",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.2789,
            0.6493
          ],
          "bRange": [
            0.862,
            1
          ]
        },
        {
          "metric": "Pass rate on eight hard tasks (Lenient (format misses counted))",
          "aValue": 0.6667,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "67% (16/24)",
          "bDisplay": "100% (24/24)",
          "winner": "b",
          "basis": "The 95% intervals do not overlap (Claude Haiku 4.5 47% to 82%; Claude Opus 5.5 86% to 100%).",
          "studySlug": "hard-model-head-to-head",
          "n": 24,
          "chartId": "hard-h2h-pass-rate",
          "aN": 24,
          "bN": 24,
          "aContext": "Claude Code · eight hard validated tasks",
          "bContext": "Claude Code · eight hard validated tasks",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.4671,
            0.8203
          ],
          "bRange": [
            0.862,
            1
          ]
        },
        {
          "metric": "Total time per call on hard tasks (separate batches)",
          "aValue": 39.01,
          "bValue": 9.18,
          "unit": "seconds",
          "aDisplay": "39.0 s",
          "bDisplay": "9.18 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Haiku 4.5 15.3 s to 75.1 s; Claude Opus 5.5 4.24 s to 27.2 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "hard-model-head-to-head",
          "n": 24,
          "chartId": "hard-h2h-total-latency",
          "aN": 24,
          "bN": 24,
          "aContext": "Claude Code · eight hard validated tasks",
          "bContext": "Claude Code · eight hard validated tasks",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            15.27,
            75.13
          ],
          "bRange": [
            4.24,
            27.21
          ]
        },
        {
          "metric": "Time to first useful output on hard tasks",
          "aValue": 35.54,
          "bValue": 6.78,
          "unit": "seconds",
          "aDisplay": "35.5 s",
          "bDisplay": "6.78 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Haiku 4.5 12.9 s to 70.3 s; Claude Opus 5.5 2.39 s to 21.8 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "hard-model-head-to-head",
          "n": 24,
          "chartId": "hard-h2h-first-useful-latency",
          "aN": 24,
          "bN": 24,
          "aContext": "Claude Code · eight hard validated tasks",
          "bContext": "Claude Code · eight hard validated tasks",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            12.88,
            70.31
          ],
          "bRange": [
            2.39,
            21.77
          ]
        },
        {
          "metric": "Output tokens per call on hard tasks (Output tokens)",
          "aValue": 5064,
          "bValue": 945,
          "unit": "tokens",
          "aDisplay": "5,064",
          "bDisplay": "945",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "hard-model-head-to-head",
          "n": 24,
          "chartId": "hard-h2h-output-tokens",
          "aN": 24,
          "bN": 24,
          "aContext": "Claude Code · eight hard validated tasks",
          "bContext": "Claude Code · eight hard validated tasks"
        },
        {
          "metric": "List-price cost per strict pass on hard tasks (calculation)",
          "aValue": 0.0672,
          "bValue": 0.02824,
          "unit": "usd",
          "aDisplay": "$0.067",
          "bDisplay": "$0.028",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.067 vs $0.028, 2.4x) is not tested against run-to-run variation.",
          "studySlug": "hard-model-head-to-head",
          "n": 24,
          "chartId": "hard-h2h-cost-per-pass",
          "aN": 24,
          "bN": 24,
          "aContext": "Claude Code · eight hard validated tasks",
          "bContext": "Claude Code · eight hard validated tasks",
          "calculation": true
        },
        {
          "metric": "Reasoning share of output tokens per call on hard tasks (calculation)",
          "aValue": 91.68,
          "bValue": 54.79,
          "unit": "percent",
          "aDisplay": "91.7%",
          "bDisplay": "54.8%",
          "winner": "unclear",
          "basis": "More or fewer percent is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "n": 24,
          "chartId": "thinking-bill-share",
          "aN": 24,
          "bN": 24,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            76.46,
            99.27
          ],
          "bRange": [
            29.92,
            95.6
          ],
          "calculation": true
        },
        {
          "metric": "List-price cost per call: reasoning, remaining output and input (calculation) (Reasoning (output tokens))",
          "aValue": 0.024492,
          "bValue": 0.012528,
          "unit": "usd",
          "aDisplay": "$0.024",
          "bDisplay": "$0.013",
          "winner": "unclear",
          "basis": "More or fewer usd is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "n": 24,
          "chartId": "thinking-bill-cost-per-call",
          "aN": 24,
          "bN": 24,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "calculation": true
        },
        {
          "metric": "List-price cost per call: reasoning, remaining output and input (calculation) (Remaining output (visible-answer estimate))",
          "aValue": 0.001795,
          "bValue": 0.008003,
          "unit": "usd",
          "aDisplay": "$0.0018",
          "bDisplay": "$0.0080",
          "winner": "unclear",
          "basis": "More or fewer usd is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "n": 24,
          "chartId": "thinking-bill-cost-per-call",
          "aN": 24,
          "bN": 24,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "calculation": true
        },
        {
          "metric": "List-price cost per call: reasoning, remaining output and input (calculation) (Input (prompt, cache priced))",
          "aValue": 0.00451,
          "bValue": 0.00771,
          "unit": "usd",
          "aDisplay": "$0.0045",
          "bDisplay": "$0.0077",
          "winner": "unclear",
          "basis": "More or fewer usd is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "n": 24,
          "chartId": "thinking-bill-cost-per-call",
          "aN": 24,
          "bN": 24,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "calculation": true
        },
        {
          "metric": "Reasoning share on short tasks vs hard tasks (calculation) (Eight hard tasks)",
          "aValue": 91.68,
          "bValue": 54.79,
          "unit": "percent",
          "aDisplay": "91.7%",
          "bDisplay": "54.8%",
          "winner": "unclear",
          "basis": "More or fewer percent is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "n": 24,
          "chartId": "thinking-bill-short-vs-hard",
          "aN": 24,
          "bN": 24,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            76.46,
            99.27
          ],
          "bRange": [
            29.92,
            95.6
          ],
          "calculation": true
        },
        {
          "metric": "Reasoning share on short tasks vs hard tasks (calculation) (Five short tasks)",
          "aValue": 90.19,
          "bValue": 0,
          "unit": "percent",
          "aDisplay": "90.2%",
          "bDisplay": "0%",
          "winner": "unclear",
          "basis": "More or fewer percent is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "n": 15,
          "chartId": "thinking-bill-short-vs-hard",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            73.1,
            97.59
          ],
          "bRange": [
            0,
            93.33
          ],
          "calculation": true
        },
        {
          "metric": "Time to first text: a 250-line answer, six models",
          "aValue": 4,
          "bValue": 1.97,
          "unit": "seconds",
          "aDisplay": "4.00 s",
          "bDisplay": "1.97 s",
          "winner": "unclear",
          "basis": "Only 4 runs per side; the run ranges (fastest to slowest) do not overlap (Claude Haiku 4.5 2.84 s to 6.38 s; Claude Opus 5.5 1.70 s to 2.35 s), but 4 runs cannot show a reliable difference. A range is not a confidence interval.",
          "studySlug": "llm-speed-anatomy",
          "n": 4,
          "chartId": "speed-anatomy-first-text",
          "aN": 4,
          "bN": 4,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            2.84,
            6.38
          ],
          "bRange": [
            1.7,
            2.35
          ]
        },
        {
          "metric": "Output speed after the first text: visible tokens per second (calculation)",
          "aValue": 153.2,
          "bValue": 155.5,
          "unit": "tokens",
          "aDisplay": "153",
          "bDisplay": "156",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "llm-speed-anatomy",
          "n": 4,
          "chartId": "speed-anatomy-output-speed",
          "aN": 4,
          "bN": 4,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            152.6,
            216.1
          ],
          "bRange": [
            154.6,
            156.4
          ],
          "calculation": true
        },
        {
          "metric": "Output speed in characters per second after the first text (calculation)",
          "aValue": 547,
          "bValue": 347,
          "unit": "count",
          "aDisplay": "547",
          "bDisplay": "347",
          "winner": "unclear",
          "basis": "More or fewer count is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "llm-speed-anatomy",
          "chartId": "speed-anatomy-chars-per-second",
          "aN": 3,
          "bN": 4,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            546,
            548
          ],
          "bRange": [
            345,
            349
          ],
          "calculation": true
        },
        {
          "metric": "Time to first text as the prompt grows: 1k",
          "aValue": 1.93,
          "bValue": 1.51,
          "unit": "seconds",
          "aDisplay": "1.93 s",
          "bDisplay": "1.51 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Haiku 4.5 1.85 s to 2.04 s; Claude Opus 5.5 1.46 s to 2.01 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "llm-speed-anatomy",
          "n": 3,
          "chartId": "speed-anatomy-prompt-size",
          "aN": 3,
          "bN": 3,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            1.85,
            2.04
          ],
          "bRange": [
            1.46,
            2.01
          ],
          "calculation": true
        },
        {
          "metric": "Time to first text as the prompt grows: 16k",
          "aValue": 2.27,
          "bValue": 1.74,
          "unit": "seconds",
          "aDisplay": "2.27 s",
          "bDisplay": "1.74 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Haiku 4.5 2.22 s to 2.47 s; Claude Opus 5.5 1.70 s to 2.97 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "llm-speed-anatomy",
          "n": 3,
          "chartId": "speed-anatomy-prompt-size",
          "aN": 3,
          "bN": 3,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            2.22,
            2.47
          ],
          "bRange": [
            1.7,
            2.97
          ],
          "calculation": true
        },
        {
          "metric": "Time to first text as the prompt grows: 64k",
          "aValue": 2.78,
          "bValue": 1.79,
          "unit": "seconds",
          "aDisplay": "2.78 s",
          "bDisplay": "1.79 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Haiku 4.5 2.45 s to 2.89 s; Claude Opus 5.5 1.72 s to 3.72 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "llm-speed-anatomy",
          "n": 3,
          "chartId": "speed-anatomy-prompt-size",
          "aN": 3,
          "bN": 3,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            2.45,
            2.89
          ],
          "bRange": [
            1.72,
            3.72
          ],
          "calculation": true
        },
        {
          "metric": "Total time per call by prompt size (1k prompt)",
          "aValue": 2.34,
          "bValue": 1.83,
          "unit": "seconds",
          "aDisplay": "2.34 s",
          "bDisplay": "1.83 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Haiku 4.5 2.22 s to 2.46 s; Claude Opus 5.5 1.82 s to 2.41 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "llm-speed-anatomy",
          "n": 3,
          "chartId": "speed-anatomy-total-by-size",
          "aN": 3,
          "bN": 3,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            2.22,
            2.46
          ],
          "bRange": [
            1.82,
            2.41
          ]
        },
        {
          "metric": "Total time per call by prompt size (16k prompt)",
          "aValue": 2.79,
          "bValue": 2.36,
          "unit": "seconds",
          "aDisplay": "2.79 s",
          "bDisplay": "2.36 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Haiku 4.5 2.58 s to 2.84 s; Claude Opus 5.5 2.11 s to 3.40 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "llm-speed-anatomy",
          "n": 3,
          "chartId": "speed-anatomy-total-by-size",
          "aN": 3,
          "bN": 3,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            2.58,
            2.84
          ],
          "bRange": [
            2.11,
            3.4
          ]
        },
        {
          "metric": "Total time per call by prompt size (64k prompt)",
          "aValue": 3.13,
          "bValue": 2.35,
          "unit": "seconds",
          "aDisplay": "3.13 s",
          "bDisplay": "2.35 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Haiku 4.5 2.84 s to 3.28 s; Claude Opus 5.5 2.26 s to 4.29 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "llm-speed-anatomy",
          "n": 3,
          "chartId": "speed-anatomy-total-by-size",
          "aN": 3,
          "bN": 3,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            2.84,
            3.28
          ],
          "bRange": [
            2.26,
            4.29
          ]
        },
        {
          "metric": "Exact lookup answers at the 1k, 16k and 64k prompt-size targets",
          "aValue": 1,
          "bValue": 0.5556,
          "unit": "rate",
          "aDisplay": "100% (9/9)",
          "bDisplay": "56% (5/9)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Haiku 4.5 70% to 100%; Claude Opus 5.5 27% to 81%), so this sample cannot separate them.",
          "studySlug": "llm-speed-anatomy",
          "n": 9,
          "chartId": "speed-anatomy-lookup-correct",
          "aN": 9,
          "bN": 9,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.7009,
            1
          ],
          "bRange": [
            0.2667,
            0.8112
          ]
        },
        {
          "metric": "Pass rate on 4 harder tasks (Strict pass)",
          "aValue": 0,
          "bValue": 0.4167,
          "unit": "rate",
          "aDisplay": "0% (0/12)",
          "bDisplay": "42% (5/12)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Haiku 4.5 0% to 24%; Claude Opus 5.5 19% to 68%), so this sample cannot separate them.",
          "studySlug": "harder-tasks-head-to-head",
          "n": 12,
          "chartId": "harder-h2h-pass-rate",
          "aN": 12,
          "bN": 12,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0,
            0.2425
          ],
          "bRange": [
            0.1933,
            0.6805
          ]
        },
        {
          "metric": "Pass rate on 4 harder tasks (Lenient (format misses counted))",
          "aValue": 0,
          "bValue": 0.5,
          "unit": "rate",
          "aDisplay": "0% (0/12)",
          "bDisplay": "50% (6/12)",
          "winner": "b",
          "basis": "The 95% intervals do not overlap (Claude Haiku 4.5 0% to 24%; Claude Opus 5.5 25% to 75%).",
          "studySlug": "harder-tasks-head-to-head",
          "n": 12,
          "chartId": "harder-h2h-pass-rate",
          "aN": 12,
          "bN": 12,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0,
            0.2425
          ],
          "bRange": [
            0.2538,
            0.7462
          ]
        },
        {
          "metric": "Calls that tried a tool although tools were off",
          "aValue": 0.0833,
          "bValue": 0.4167,
          "unit": "rate",
          "aDisplay": "8% (1/12)",
          "bDisplay": "42% (5/12)",
          "winner": "unclear",
          "basis": "More or fewer rate is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "harder-tasks-head-to-head",
          "n": 12,
          "chartId": "harder-h2h-tool-attempts",
          "aN": 12,
          "bN": 12,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.0149,
            0.3539
          ],
          "bRange": [
            0.1933,
            0.6805
          ]
        },
        {
          "metric": "Strict pass rate by task: 10x10 nonogram",
          "aValue": 0,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "0% (0/3)",
          "bDisplay": "100% (3/3)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Haiku 4.5 0% to 56%; Claude Opus 5.5 44% to 100%), so this sample cannot separate them.",
          "studySlug": "harder-tasks-head-to-head",
          "n": 3,
          "chartId": "harder-h2h-pass-by-task",
          "aN": 3,
          "bN": 3,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0,
            0.5615
          ],
          "bRange": [
            0.4385,
            1
          ]
        },
        {
          "metric": "Strict pass rate by task: Sudoku, 22 givens",
          "aValue": 0,
          "bValue": 0,
          "unit": "rate",
          "aDisplay": "0% (0/3)",
          "bDisplay": "0% (0/3)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Haiku 4.5 0% to 56%; Claude Opus 5.5 0% to 56%), so this sample cannot separate them.",
          "studySlug": "harder-tasks-head-to-head",
          "n": 3,
          "chartId": "harder-h2h-pass-by-task",
          "aN": 3,
          "bN": 3,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0,
            0.5615
          ],
          "bRange": [
            0,
            0.5615
          ]
        },
        {
          "metric": "Strict pass rate by task: 6x6 Skyscrapers",
          "aValue": 0,
          "bValue": 0.3333,
          "unit": "rate",
          "aDisplay": "0% (0/3)",
          "bDisplay": "33% (1/3)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Haiku 4.5 0% to 56%; Claude Opus 5.5 6% to 79%), so this sample cannot separate them.",
          "studySlug": "harder-tasks-head-to-head",
          "n": 3,
          "chartId": "harder-h2h-pass-by-task",
          "aN": 3,
          "bN": 3,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0,
            0.5615
          ],
          "bRange": [
            0.0615,
            0.7923
          ]
        },
        {
          "metric": "Strict pass rate by task: Seeded shuffle output",
          "aValue": 0,
          "bValue": 0.3333,
          "unit": "rate",
          "aDisplay": "0% (0/3)",
          "bDisplay": "33% (1/3)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Haiku 4.5 0% to 56%; Claude Opus 5.5 6% to 79%), so this sample cannot separate them.",
          "studySlug": "harder-tasks-head-to-head",
          "n": 3,
          "chartId": "harder-h2h-pass-by-task",
          "aN": 3,
          "bN": 3,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0,
            0.5615
          ],
          "bRange": [
            0.0615,
            0.7923
          ]
        },
        {
          "metric": "Total time per call on harder tasks",
          "aValue": 108.98,
          "bValue": 80.34,
          "unit": "seconds",
          "aDisplay": "109.0 s",
          "bDisplay": "80.3 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Haiku 4.5 25.7 s to 223.9 s; Claude Opus 5.5 3.82 s to 279.5 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-total-latency",
          "aN": 10,
          "bN": 9,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            25.73,
            223.95
          ],
          "bRange": [
            3.82,
            279.5
          ]
        },
        {
          "metric": "Output tokens per call on harder tasks (Output tokens)",
          "aValue": 12508,
          "bValue": 8420,
          "unit": "tokens",
          "aDisplay": "12,508",
          "bDisplay": "8,420",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-output-tokens",
          "aN": 10,
          "bN": 9,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            2965,
            26532
          ],
          "bRange": [
            279,
            40044
          ]
        }
      ]
    },
    {
      "slug": "claude-code-cli-vs-codex-cli",
      "a": "claude-code-cli",
      "b": "codex-cli",
      "title": "Claude Code vs Codex CLI",
      "seoTitle": "Claude Code vs Codex CLI: measured benchmarks",
      "description": "Claude Code vs Codex CLI: 66 measured metrics from 11 studies (Pass rate on five validated tasks; Total time per call; more), with sample sizes and intervals.",
      "verdict": "Claude Code and Codex CLI share 66 measured metrics and 22 list-price calculations from 12 studies. Claude Code leads on 7 rows: Time per coding session, 23.1 s vs 113.4 s; Same prompt, 10 times: time per call (Exact number), 6.89 s vs 13.4 s; Same prompt, 10 times: time per call (Code fix), 2.67 s vs 11.3 s; and 4 more. Codex CLI leads on 1 row: Strict pass rate by task: 6x6 Skyscrapers, 100% (4/4) vs 0% (0/4). On those rows the 95% intervals and run ranges do not overlap; only a 95% interval is a confidence interval. The other rows are 31 ties and 49 unclear; each row says why. Calculation rows are derived from list prices and recorded counts; they are not bills or runs. Each run pairs a CLI with a model, so these rows cannot separate the CLI from the model; the contexts name both. Some rows rest on small samples (n = 2 at the smallest).",
      "rows": [
        {
          "metric": "Pass rate on five validated tasks",
          "aValue": 0.8,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "80% (12/15)",
          "bDisplay": "100% (15/15)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Code 55% to 93%; Codex CLI 80% to 100%), so this sample cannot separate them.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-pass-rate",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Sonnet 5.5 · five short validated tasks",
          "bContext": "GPT-6.1 Sol · effort medium · five short validated tasks",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.5481,
            0.9295
          ],
          "bRange": [
            0.7961,
            1
          ]
        },
        {
          "metric": "Total time per call",
          "aValue": 2.31,
          "bValue": 5.65,
          "unit": "seconds",
          "aDisplay": "2.31 s",
          "bDisplay": "5.65 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Code 2.17 s to 7.73 s; Codex CLI 4.10 s to 25.5 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-total-latency",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Sonnet 5.5 · five short validated tasks",
          "bContext": "GPT-6.1 Sol · effort medium · five short validated tasks",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            2.17,
            7.73
          ],
          "bRange": [
            4.1,
            25.46
          ]
        },
        {
          "metric": "Time to first useful output",
          "aValue": 1.56,
          "bValue": 5.05,
          "unit": "seconds",
          "aDisplay": "1.56 s",
          "bDisplay": "5.05 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Code 0.99 s to 6.39 s; Codex CLI 3.36 s to 17.8 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-first-useful-latency",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Sonnet 5.5 · five short validated tasks",
          "bContext": "GPT-6.1 Sol · effort medium · five short validated tasks",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            0.99,
            6.39
          ],
          "bRange": [
            3.36,
            17.82
          ]
        },
        {
          "metric": "Input tokens per call: what the CLI sends (Cache read)",
          "aValue": 1401,
          "bValue": 5180,
          "unit": "tokens",
          "aDisplay": "1,401",
          "bDisplay": "5,180",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-input-tokens",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Sonnet 5.5 · five short validated tasks",
          "bContext": "GPT-6.1 Sol · effort medium · five short validated tasks"
        },
        {
          "metric": "Input tokens per call: what the CLI sends (Other input)",
          "aValue": 685,
          "bValue": 6943,
          "unit": "tokens",
          "aDisplay": "685",
          "bDisplay": "6,943",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-input-tokens",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Sonnet 5.5 · five short validated tasks",
          "bContext": "GPT-6.1 Sol · effort medium · five short validated tasks"
        },
        {
          "metric": "Output tokens per call (Output tokens)",
          "aValue": 107,
          "bValue": 42,
          "unit": "tokens",
          "aDisplay": "107",
          "bDisplay": "42",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-output-tokens",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Sonnet 5.5 · five short validated tasks",
          "bContext": "GPT-6.1 Sol · effort medium · five short validated tasks"
        },
        {
          "metric": "List-price cost per call (calculation)",
          "aValue": 0.0036,
          "bValue": 0.01018,
          "unit": "usd",
          "aDisplay": "$0.0036",
          "bDisplay": "$0.010",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Code $0.0034 to $0.010; Codex CLI $0.0054 to $0.027); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-list-price-per-call",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Sonnet 5.5 · five short validated tasks",
          "bContext": "GPT-6.1 Sol · effort medium · five short validated tasks",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            0.00342,
            0.01021
          ],
          "bRange": [
            0.0054,
            0.02686
          ],
          "calculation": true
        },
        {
          "metric": "List-price cost per passing answer (calculation)",
          "aValue": 0.00624,
          "bValue": 0.01564,
          "unit": "usd",
          "aDisplay": "$0.0062",
          "bDisplay": "$0.016",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.0062 vs $0.016, 2.5x) is not tested against run-to-run variation.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-cost-per-pass",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Sonnet 5.5 · five short validated tasks",
          "bContext": "GPT-6.1 Sol · effort medium · five short validated tasks",
          "calculation": true
        },
        {
          "metric": "Pass rate on eight hard tasks (Strict pass)",
          "aValue": 1,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "100% (24/24)",
          "bDisplay": "100% (16/16)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Code 86% to 100%; Codex CLI 81% to 100%), so this sample cannot separate them.",
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-pass-rate",
          "aN": 24,
          "bN": 16,
          "aContext": "Claude Sonnet 5.5 · eight hard validated tasks",
          "bContext": "GPT-6.1 Sol · effort medium · eight hard validated tasks",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.862,
            1
          ],
          "bRange": [
            0.8064,
            1
          ]
        },
        {
          "metric": "Pass rate on eight hard tasks (Lenient (format misses counted))",
          "aValue": 1,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "100% (24/24)",
          "bDisplay": "100% (16/16)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Code 86% to 100%; Codex CLI 81% to 100%), so this sample cannot separate them.",
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-pass-rate",
          "aN": 24,
          "bN": 16,
          "aContext": "Claude Sonnet 5.5 · eight hard validated tasks",
          "bContext": "GPT-6.1 Sol · effort medium · eight hard validated tasks",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.862,
            1
          ],
          "bRange": [
            0.8064,
            1
          ]
        },
        {
          "metric": "Total time per call on hard tasks (separate batches)",
          "aValue": 7.75,
          "bValue": 13.11,
          "unit": "seconds",
          "aDisplay": "7.75 s",
          "bDisplay": "13.1 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Code 2.26 s to 34.8 s; Codex CLI 8.54 s to 61.6 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-total-latency",
          "aN": 24,
          "bN": 16,
          "aContext": "Claude Sonnet 5.5 · eight hard validated tasks",
          "bContext": "GPT-6.1 Sol · effort medium · eight hard validated tasks",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            2.26,
            34.79
          ],
          "bRange": [
            8.54,
            61.6
          ]
        },
        {
          "metric": "Time to first useful output on hard tasks",
          "aValue": 5.95,
          "bValue": 10.23,
          "unit": "seconds",
          "aDisplay": "5.95 s",
          "bDisplay": "10.2 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Code 0.86 s to 30.6 s; Codex CLI 6.09 s to 40.4 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-first-useful-latency",
          "aN": 24,
          "bN": 16,
          "aContext": "Claude Sonnet 5.5 · eight hard validated tasks",
          "bContext": "GPT-6.1 Sol · effort medium · eight hard validated tasks",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            0.86,
            30.57
          ],
          "bRange": [
            6.09,
            40.41
          ]
        },
        {
          "metric": "Output tokens per call on hard tasks (Output tokens)",
          "aValue": 1050,
          "bValue": 335,
          "unit": "tokens",
          "aDisplay": "1,050",
          "bDisplay": "335",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-output-tokens",
          "aN": 24,
          "bN": 16,
          "aContext": "Claude Sonnet 5.5 · eight hard validated tasks",
          "bContext": "GPT-6.1 Sol · effort medium · eight hard validated tasks"
        },
        {
          "metric": "List-price cost per strict pass on hard tasks (calculation)",
          "aValue": 0.01435,
          "bValue": 0.02564,
          "unit": "usd",
          "aDisplay": "$0.014",
          "bDisplay": "$0.026",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.014 vs $0.026) is not tested against run-to-run variation.",
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-cost-per-pass",
          "aN": 24,
          "bN": 16,
          "aContext": "Claude Sonnet 5.5 · eight hard validated tasks",
          "bContext": "GPT-6.1 Sol · effort medium · eight hard validated tasks",
          "calculation": true
        },
        {
          "metric": "Coding sessions that passed every hidden check",
          "aValue": 1,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "100% (12/12)",
          "bDisplay": "100% (12/12)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Code 76% to 100%; Codex CLI 76% to 100%), so this sample cannot separate them.",
          "studySlug": "coding-agents-head-to-head",
          "n": 12,
          "chartId": "coding-agents-pass-rate",
          "aN": 12,
          "bN": 12,
          "aContext": "Claude Sonnet 5.5 · six small repository tasks with hidden tests",
          "bContext": "GPT-6.1 Sol · effort medium · tester’s AGENTS.md · six small repository tasks with hidden tests",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.7575,
            1
          ],
          "bRange": [
            0.7575,
            1
          ]
        },
        {
          "metric": "Time per coding session",
          "aValue": 23.1,
          "bValue": 113.4,
          "unit": "seconds",
          "aDisplay": "23.1 s",
          "bDisplay": "113.4 s",
          "winner": "a",
          "basis": "The run ranges (fastest to slowest) do not overlap (Claude Code 18.7 s to 44.5 s; Codex CLI 78.5 s to 221.9 s). A range is not a confidence interval.",
          "studySlug": "coding-agents-head-to-head",
          "n": 12,
          "chartId": "coding-agents-wall-time",
          "aN": 12,
          "bN": 12,
          "aContext": "Claude Sonnet 5.5 · six small repository tasks with hidden tests",
          "bContext": "GPT-6.1 Sol · effort medium · tester’s AGENTS.md · six small repository tasks with hidden tests",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            18.7,
            44.5
          ],
          "bRange": [
            78.5,
            221.9
          ]
        },
        {
          "metric": "Tool calls per coding session",
          "aValue": 7.5,
          "bValue": 12.5,
          "unit": "calls",
          "aDisplay": "7.5",
          "bDisplay": "12.5",
          "winner": "unclear",
          "basis": "More or fewer calls is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "coding-agents-head-to-head",
          "n": 12,
          "chartId": "coding-agents-tool-calls",
          "aN": 12,
          "bN": 12,
          "aContext": "Claude Sonnet 5.5 · six small repository tasks with hidden tests",
          "bContext": "GPT-6.1 Sol · effort medium · tester’s AGENTS.md · six small repository tasks with hidden tests",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            3,
            14
          ],
          "bRange": [
            8,
            18
          ]
        },
        {
          "metric": "List-price cost per passing coding session (calculation)",
          "aValue": 0.085,
          "bValue": 0.0978,
          "unit": "usd",
          "aDisplay": "$0.085",
          "bDisplay": "$0.098",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.085 vs $0.098) is not tested against run-to-run variation.",
          "studySlug": "coding-agents-head-to-head",
          "n": 12,
          "chartId": "coding-agents-cost-per-pass",
          "aN": 12,
          "bN": 12,
          "aContext": "Claude Sonnet 5.5 · six small repository tasks with hidden tests",
          "bContext": "GPT-6.1 Sol · effort medium · tester’s AGENTS.md · six small repository tasks with hidden tests",
          "calculation": true
        },
        {
          "metric": "Strict pass rate by effort on eight hard tasks",
          "aValue": 1,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "100% (16/16)",
          "bDisplay": "100% (16/16)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Code 81% to 100%; Codex CLI 81% to 100%), so this sample cannot separate them.",
          "studySlug": "effort-ladder",
          "n": 16,
          "chartId": "effort-ladder-pass-rate",
          "aN": 16,
          "bN": 16,
          "aContext": "Claude Sonnet 5.5 · effort medium · eight hard validated tasks, effort ladder",
          "bContext": "GPT-6.1 Sol · effort medium · eight hard validated tasks, effort ladder",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.8064,
            1
          ],
          "bRange": [
            0.8064,
            1
          ]
        },
        {
          "metric": "Total time per call by effort on hard tasks",
          "aValue": 7.63,
          "bValue": 13.11,
          "unit": "seconds",
          "aDisplay": "7.63 s",
          "bDisplay": "13.1 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Code 2.71 s to 24.0 s; Codex CLI 8.54 s to 61.6 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "effort-ladder",
          "n": 16,
          "chartId": "effort-ladder-total-latency",
          "aN": 16,
          "bN": 16,
          "aContext": "Claude Sonnet 5.5 · effort medium · eight hard validated tasks, effort ladder",
          "bContext": "GPT-6.1 Sol · effort medium · eight hard validated tasks, effort ladder",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            2.71,
            24.01
          ],
          "bRange": [
            8.54,
            61.6
          ]
        },
        {
          "metric": "Output tokens per call by effort on hard tasks (Output tokens)",
          "aValue": 770,
          "bValue": 335,
          "unit": "tokens",
          "aDisplay": "770",
          "bDisplay": "335",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "effort-ladder",
          "n": 16,
          "chartId": "effort-ladder-output-tokens",
          "aN": 16,
          "bN": 16,
          "aContext": "Claude Sonnet 5.5 · effort medium · eight hard validated tasks, effort ladder",
          "bContext": "GPT-6.1 Sol · effort medium · eight hard validated tasks, effort ladder"
        },
        {
          "metric": "List-price cost per strict pass by effort (calculation)",
          "aValue": 0.01352,
          "bValue": 0.02564,
          "unit": "usd",
          "aDisplay": "$0.014",
          "bDisplay": "$0.026",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.014 vs $0.026) is not tested against run-to-run variation.",
          "studySlug": "effort-ladder",
          "n": 16,
          "chartId": "effort-ladder-cost-per-pass",
          "aN": 16,
          "bN": 16,
          "aContext": "Claude Sonnet 5.5 · effort medium · eight hard validated tasks, effort ladder",
          "bContext": "GPT-6.1 Sol · effort medium · eight hard validated tasks, effort ladder",
          "calculation": true
        },
        {
          "metric": "Same prompt, 10 times: strict pass rate (Exact number)",
          "aValue": 1,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "100% (10/10)",
          "bDisplay": "100% (10/10)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Code 72% to 100%; Codex CLI 72% to 100%), so this sample cannot separate them.",
          "studySlug": "caching-consistency",
          "n": 10,
          "chartId": "consistency-pass-rate",
          "aN": 10,
          "bN": 10,
          "aContext": "Claude Sonnet 5.5 · same prompt repeated 10 times",
          "bContext": "GPT-6.1 Sol · effort medium · same prompt repeated 10 times",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.7225,
            1
          ],
          "bRange": [
            0.7225,
            1
          ]
        },
        {
          "metric": "Same prompt, 10 times: strict pass rate (JSON object)",
          "aValue": 1,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "100% (10/10)",
          "bDisplay": "100% (10/10)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Code 72% to 100%; Codex CLI 72% to 100%), so this sample cannot separate them.",
          "studySlug": "caching-consistency",
          "n": 10,
          "chartId": "consistency-pass-rate",
          "aN": 10,
          "bN": 10,
          "aContext": "Claude Sonnet 5.5 · same prompt repeated 10 times",
          "bContext": "GPT-6.1 Sol · effort medium · same prompt repeated 10 times",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.7225,
            1
          ],
          "bRange": [
            0.7225,
            1
          ]
        },
        {
          "metric": "Same prompt, 10 times: strict pass rate (Code fix)",
          "aValue": 1,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "100% (10/10)",
          "bDisplay": "100% (10/10)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Code 72% to 100%; Codex CLI 72% to 100%), so this sample cannot separate them.",
          "studySlug": "caching-consistency",
          "n": 10,
          "chartId": "consistency-pass-rate",
          "aN": 10,
          "bN": 10,
          "aContext": "Claude Sonnet 5.5 · same prompt repeated 10 times",
          "bContext": "GPT-6.1 Sol · effort medium · same prompt repeated 10 times",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.7225,
            1
          ],
          "bRange": [
            0.7225,
            1
          ]
        },
        {
          "metric": "Same prompt, 10 times: how many different answers (Exact number)",
          "aValue": 1,
          "bValue": 1,
          "unit": "count",
          "aDisplay": "1",
          "bDisplay": "1",
          "winner": "tie",
          "basis": "Same value. More or fewer is not better by itself for this metric.",
          "studySlug": "caching-consistency",
          "n": 10,
          "chartId": "consistency-distinct-answers",
          "aN": 10,
          "bN": 10,
          "aContext": "Claude Sonnet 5.5 · same prompt repeated 10 times",
          "bContext": "GPT-6.1 Sol · effort medium · same prompt repeated 10 times"
        },
        {
          "metric": "Same prompt, 10 times: how many different answers (JSON object)",
          "aValue": 1,
          "bValue": 1,
          "unit": "count",
          "aDisplay": "1",
          "bDisplay": "1",
          "winner": "tie",
          "basis": "Same value. More or fewer is not better by itself for this metric.",
          "studySlug": "caching-consistency",
          "n": 10,
          "chartId": "consistency-distinct-answers",
          "aN": 10,
          "bN": 10,
          "aContext": "Claude Sonnet 5.5 · same prompt repeated 10 times",
          "bContext": "GPT-6.1 Sol · effort medium · same prompt repeated 10 times"
        },
        {
          "metric": "Same prompt, 10 times: how many different answers (Code fix)",
          "aValue": 3,
          "bValue": 6,
          "unit": "count",
          "aDisplay": "3",
          "bDisplay": "6",
          "winner": "unclear",
          "basis": "More or fewer count is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "caching-consistency",
          "n": 10,
          "chartId": "consistency-distinct-answers",
          "aN": 10,
          "bN": 10,
          "aContext": "Claude Sonnet 5.5 · same prompt repeated 10 times",
          "bContext": "GPT-6.1 Sol · effort medium · same prompt repeated 10 times"
        },
        {
          "metric": "Same prompt, 10 times: time per call (Exact number)",
          "aValue": 6.89,
          "bValue": 13.38,
          "unit": "seconds",
          "aDisplay": "6.89 s",
          "bDisplay": "13.4 s",
          "winner": "a",
          "basis": "The run ranges (fastest to slowest) do not overlap (Claude Code 5.81 s to 7.81 s; Codex CLI 12.3 s to 18.0 s). A range is not a confidence interval.",
          "studySlug": "caching-consistency",
          "n": 10,
          "chartId": "consistency-latency-spread",
          "aN": 10,
          "bN": 10,
          "aContext": "Claude Sonnet 5.5 · same prompt repeated 10 times",
          "bContext": "GPT-6.1 Sol · effort medium · same prompt repeated 10 times",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            5.81,
            7.81
          ],
          "bRange": [
            12.29,
            17.97
          ]
        },
        {
          "metric": "Same prompt, 10 times: time per call (JSON object)",
          "aValue": 2.89,
          "bValue": 6.42,
          "unit": "seconds",
          "aDisplay": "2.89 s",
          "bDisplay": "6.42 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Code 2.68 s to 5.30 s; Codex CLI 5.25 s to 8.26 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "caching-consistency",
          "n": 10,
          "chartId": "consistency-latency-spread",
          "aN": 10,
          "bN": 10,
          "aContext": "Claude Sonnet 5.5 · same prompt repeated 10 times",
          "bContext": "GPT-6.1 Sol · effort medium · same prompt repeated 10 times",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            2.68,
            5.3
          ],
          "bRange": [
            5.25,
            8.26
          ]
        },
        {
          "metric": "Same prompt, 10 times: time per call (Code fix)",
          "aValue": 2.67,
          "bValue": 11.29,
          "unit": "seconds",
          "aDisplay": "2.67 s",
          "bDisplay": "11.3 s",
          "winner": "a",
          "basis": "The run ranges (fastest to slowest) do not overlap (Claude Code 2.32 s to 4.34 s; Codex CLI 9.08 s to 14.8 s). A range is not a confidence interval.",
          "studySlug": "caching-consistency",
          "n": 10,
          "chartId": "consistency-latency-spread",
          "aN": 10,
          "bN": 10,
          "aContext": "Claude Sonnet 5.5 · same prompt repeated 10 times",
          "bContext": "GPT-6.1 Sol · effort medium · same prompt repeated 10 times",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            2.32,
            4.34
          ],
          "bRange": [
            9.08,
            14.85
          ]
        },
        {
          "metric": "CLI start-up tax on a one-word answer (First output event)",
          "aValue": 563,
          "bValue": 489,
          "unit": "ms",
          "aDisplay": "563 ms",
          "bDisplay": "489 ms",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Code 519 ms to 726 ms; Codex CLI 354 ms to 1,304 ms); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "routing-overhead",
          "n": 5,
          "chartId": "cli-startup-tax",
          "aN": 5,
          "bN": 5,
          "aContext": "Claude Haiku 4.5 · CLI start-up, one-word prompt, 5 runs",
          "bContext": "default model · CLI start-up, one-word prompt, 5 runs",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            519,
            726
          ],
          "bRange": [
            354,
            1304
          ]
        },
        {
          "metric": "CLI start-up tax on a one-word answer (First model output)",
          "aValue": 1461,
          "bValue": 5059,
          "unit": "ms",
          "aDisplay": "1,461 ms",
          "bDisplay": "5,059 ms",
          "winner": "a",
          "basis": "The run ranges (fastest to slowest) do not overlap (Claude Code 1,206 ms to 2,308 ms; Codex CLI 4,391 ms to 5,478 ms). A range is not a confidence interval. Samples are small (5 runs per side).",
          "studySlug": "routing-overhead",
          "n": 5,
          "chartId": "cli-startup-tax",
          "aN": 5,
          "bN": 5,
          "aContext": "Claude Haiku 4.5 · CLI start-up, one-word prompt, 5 runs",
          "bContext": "default model · CLI start-up, one-word prompt, 5 runs",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            1206,
            2308
          ],
          "bRange": [
            4391,
            5478
          ]
        },
        {
          "metric": "CLI start-up tax on a one-word answer (Total wall time)",
          "aValue": 2529,
          "bValue": 5999,
          "unit": "ms",
          "aDisplay": "2,529 ms",
          "bDisplay": "5,999 ms",
          "winner": "a",
          "basis": "The run ranges (fastest to slowest) do not overlap (Claude Code 2,273 ms to 3,382 ms; Codex CLI 5,367 ms to 6,506 ms). A range is not a confidence interval. Samples are small (5 runs per side).",
          "studySlug": "routing-overhead",
          "n": 5,
          "chartId": "cli-startup-tax",
          "aN": 5,
          "bN": 5,
          "aContext": "Claude Haiku 4.5 · CLI start-up, one-word prompt, 5 runs",
          "bContext": "default model · CLI start-up, one-word prompt, 5 runs",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            2273,
            3382
          ],
          "bRange": [
            5367,
            6506
          ]
        },
        {
          "metric": "Input tokens a CLI sends for a one-word answer",
          "aValue": 6761,
          "bValue": 17051,
          "unit": "tokens",
          "aDisplay": "6,761",
          "bDisplay": "17,051",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "routing-overhead",
          "n": 5,
          "chartId": "cli-startup-input-tokens",
          "aN": 5,
          "bN": 5,
          "aContext": "Claude Haiku 4.5 · CLI start-up, one-word prompt, 5 runs",
          "bContext": "default model · CLI start-up, one-word prompt, 5 runs"
        },
        {
          "metric": "Repairing a scheduler: Claude Code vs Codex vs API (Total time)",
          "aValue": 15,
          "bValue": 61.16,
          "unit": "seconds",
          "aDisplay": "15.0 s",
          "bDisplay": "61.2 s",
          "winner": "unclear",
          "basis": "Only 3 runs per side; the run ranges (fastest to slowest) do not overlap (Claude Code 13.9 s to 15.9 s; Codex CLI 59.9 s to 69.5 s), but 3 runs cannot show a reliable difference. A range is not a confidence interval.",
          "studySlug": "cli-model-latency-tokens",
          "n": 3,
          "chartId": "scheduler-repair-claude-vs-codex",
          "aN": 3,
          "bN": 3,
          "aContext": "Sonnet 5.5 · effort medium · scheduler repair, 296 checks, 3 runs",
          "bContext": "GPT-6.1 Sol · effort medium · scheduler repair, 296 checks, 3 runs",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            13.89,
            15.89
          ],
          "bRange": [
            59.9,
            69.51
          ]
        },
        {
          "metric": "Repairing a scheduler: Claude Code vs Codex vs API (First useful output)",
          "aValue": 7.55,
          "bValue": 15.56,
          "unit": "seconds",
          "aDisplay": "7.55 s",
          "bDisplay": "15.6 s",
          "winner": "unclear",
          "basis": "Only 3 runs per side; the run ranges (fastest to slowest) do not overlap (Claude Code 6.77 s to 7.63 s; Codex CLI 13.7 s to 23.0 s), but 3 runs cannot show a reliable difference. A range is not a confidence interval.",
          "studySlug": "cli-model-latency-tokens",
          "n": 3,
          "chartId": "scheduler-repair-claude-vs-codex",
          "aN": 3,
          "bN": 3,
          "aContext": "Sonnet 5.5 · effort medium · scheduler repair, 296 checks, 3 runs",
          "bContext": "GPT-6.1 Sol · effort medium · scheduler repair, 296 checks, 3 runs",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            6.77,
            7.63
          ],
          "bRange": [
            13.65,
            23.04
          ]
        },
        {
          "metric": "Output tokens to repair the scheduler (Output tokens)",
          "aValue": 2227,
          "bValue": 1181,
          "unit": "tokens",
          "aDisplay": "2,227",
          "bDisplay": "1,181",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "cli-model-latency-tokens",
          "n": 3,
          "chartId": "scheduler-repair-output-tokens",
          "aN": 3,
          "bN": 3,
          "aContext": "Sonnet 5.5 · effort medium · scheduler repair, 296 checks, 3 runs",
          "bContext": "GPT-6.1 Sol · effort medium · scheduler repair, 296 checks, 3 runs"
        },
        {
          "metric": "Strict pass rate: single call vs agent loop on eight hard tasks",
          "aValue": 1,
          "bValue": 0.625,
          "unit": "rate",
          "aDisplay": "100% (24/24)",
          "bDisplay": "63% (10/16)",
          "winner": "a",
          "basis": "The 95% intervals do not overlap (Claude Code 86% to 100%; Codex CLI 39% to 82%).",
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-pass-rate",
          "aN": 24,
          "bN": 16,
          "aContext": "Claude Sonnet 5.5 · single call",
          "bContext": "GPT-6 Luna · single call",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.862,
            1
          ],
          "bRange": [
            0.3864,
            0.8152
          ]
        },
        {
          "metric": "Strict passes per task: single call vs agent loop: Interval merge fix",
          "aValue": 1,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "100% (3/3)",
          "bDisplay": "100% (2/2)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Code 44% to 100%; Codex CLI 34% to 100%), so this sample cannot separate them.",
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "aN": 3,
          "bN": 2,
          "aContext": "Claude Sonnet 5.5 · single call",
          "bContext": "GPT-6 Luna · single call",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.4385,
            1
          ],
          "bRange": [
            0.3424,
            1
          ]
        },
        {
          "metric": "Strict passes per task: single call vs agent loop: DST day-length fix",
          "aValue": 1,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "100% (3/3)",
          "bDisplay": "100% (2/2)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Code 44% to 100%; Codex CLI 34% to 100%), so this sample cannot separate them.",
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "aN": 3,
          "bN": 2,
          "aContext": "Claude Sonnet 5.5 · single call",
          "bContext": "GPT-6 Luna · single call",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.4385,
            1
          ],
          "bRange": [
            0.3424,
            1
          ]
        },
        {
          "metric": "Strict passes per task: single call vs agent loop: CSV parser",
          "aValue": 1,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "100% (3/3)",
          "bDisplay": "100% (2/2)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Code 44% to 100%; Codex CLI 34% to 100%), so this sample cannot separate them.",
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "aN": 3,
          "bN": 2,
          "aContext": "Claude Sonnet 5.5 · single call",
          "bContext": "GPT-6 Luna · single call",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.4385,
            1
          ],
          "bRange": [
            0.3424,
            1
          ]
        },
        {
          "metric": "Strict passes per task: single call vs agent loop: Event-loop order",
          "aValue": 1,
          "bValue": 0,
          "unit": "rate",
          "aDisplay": "100% (3/3)",
          "bDisplay": "0% (0/2)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Code 44% to 100%; Codex CLI 0% to 66%), so this sample cannot separate them.",
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "aN": 3,
          "bN": 2,
          "aContext": "Claude Sonnet 5.5 · single call",
          "bContext": "GPT-6 Luna · single call",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.4385,
            1
          ],
          "bRange": [
            0,
            0.6576
          ]
        },
        {
          "metric": "Strict passes per task: single call vs agent loop: Room schedule",
          "aValue": 1,
          "bValue": 0.5,
          "unit": "rate",
          "aDisplay": "100% (3/3)",
          "bDisplay": "50% (1/2)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Code 44% to 100%; Codex CLI 9% to 91%), so this sample cannot separate them.",
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "aN": 3,
          "bN": 2,
          "aContext": "Claude Sonnet 5.5 · single call",
          "bContext": "GPT-6 Luna · single call",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.4385,
            1
          ],
          "bRange": [
            0.0945,
            0.9055
          ]
        },
        {
          "metric": "Strict passes per task: single call vs agent loop: SemVer regex",
          "aValue": 1,
          "bValue": 0.5,
          "unit": "rate",
          "aDisplay": "100% (3/3)",
          "bDisplay": "50% (1/2)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Code 44% to 100%; Codex CLI 9% to 91%), so this sample cannot separate them.",
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "aN": 3,
          "bN": 2,
          "aContext": "Claude Sonnet 5.5 · single call",
          "bContext": "GPT-6 Luna · single call",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.4385,
            1
          ],
          "bRange": [
            0.0945,
            0.9055
          ]
        },
        {
          "metric": "Strict passes per task: single call vs agent loop: Money refactor",
          "aValue": 1,
          "bValue": 0,
          "unit": "rate",
          "aDisplay": "100% (3/3)",
          "bDisplay": "0% (0/2)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Code 44% to 100%; Codex CLI 0% to 66%), so this sample cannot separate them.",
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "aN": 3,
          "bN": 2,
          "aContext": "Claude Sonnet 5.5 · single call",
          "bContext": "GPT-6 Luna · single call",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.4385,
            1
          ],
          "bRange": [
            0,
            0.6576
          ]
        },
        {
          "metric": "Strict passes per task: single call vs agent loop: SQL report",
          "aValue": 1,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "100% (3/3)",
          "bDisplay": "100% (2/2)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Code 44% to 100%; Codex CLI 34% to 100%), so this sample cannot separate them.",
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "aN": 3,
          "bN": 2,
          "aContext": "Claude Sonnet 5.5 · single call",
          "bContext": "GPT-6 Luna · single call",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.4385,
            1
          ],
          "bRange": [
            0.3424,
            1
          ]
        },
        {
          "metric": "Total time per attempt: single call vs agent loop",
          "aValue": 7.75,
          "bValue": 5.16,
          "unit": "seconds",
          "aDisplay": "7.75 s",
          "bDisplay": "5.16 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Code 2.26 s to 34.8 s; Codex CLI 3.59 s to 11.3 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-total-time",
          "aN": 24,
          "bN": 16,
          "aContext": "Claude Sonnet 5.5 · single call",
          "bContext": "GPT-6 Luna · single call",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            2.26,
            34.79
          ],
          "bRange": [
            3.59,
            11.32
          ]
        },
        {
          "metric": "Tokens per attempt: single call vs agent loop (Input tokens (cache reads included))",
          "aValue": 2281,
          "bValue": 11582,
          "unit": "tokens",
          "aDisplay": "2,281",
          "bDisplay": "11,582",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-tokens",
          "aN": 24,
          "bN": 16,
          "aContext": "Claude Sonnet 5.5 · single call",
          "bContext": "GPT-6 Luna · single call",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            2234,
            2669
          ],
          "bRange": [
            11526,
            11818
          ]
        },
        {
          "metric": "Tokens per attempt: single call vs agent loop (Output tokens)",
          "aValue": 1050,
          "bValue": 345,
          "unit": "tokens",
          "aDisplay": "1,050",
          "bDisplay": "345",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-tokens",
          "aN": 24,
          "bN": 16,
          "aContext": "Claude Sonnet 5.5 · single call",
          "bContext": "GPT-6 Luna · single call",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            176,
            3895
          ],
          "bRange": [
            36,
            634
          ]
        },
        {
          "metric": "Tool calls per agent-loop attempt",
          "aValue": 0,
          "bValue": 0,
          "unit": "count",
          "aDisplay": "0",
          "bDisplay": "0",
          "winner": "tie",
          "basis": "Same value. More or fewer is not better by itself for this metric.",
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-tool-calls",
          "aN": 16,
          "bN": 14,
          "aContext": "Claude Sonnet 5.5 · agent loop",
          "bContext": "GPT-6 Luna · agent loop",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            0,
            3
          ],
          "bRange": [
            0,
            1
          ]
        },
        {
          "metric": "List-price cost per strict pass: single call vs agent loop (calculation)",
          "aValue": 0.01435,
          "bValue": 0.00116,
          "unit": "usd",
          "aDisplay": "$0.014",
          "bDisplay": "$0.0012",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.014 vs $0.0012, 12x) is not tested against run-to-run variation.",
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-cost-per-pass",
          "aN": 24,
          "bN": 16,
          "aContext": "Claude Sonnet 5.5 · single call",
          "bContext": "GPT-6 Luna · single call",
          "calculation": true
        },
        {
          "metric": "Does a JSON schema raise the pass rate? Instructions vs schema mode (Strict pass: the whole reply is the right JSON)",
          "aValue": 1,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "100% (12/12)",
          "bDisplay": "100% (12/12)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Code 76% to 100%; Codex CLI 76% to 100%), so this sample cannot separate them.",
          "studySlug": "json-schema-vs-instructions",
          "n": 12,
          "chartId": "structured-output-pass-rate",
          "aN": 12,
          "bN": 12,
          "aContext": "Claude Sonnet 5.5 · instructions",
          "bContext": "GPT-6.1 Sol · effort low · instructions",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.7575,
            1
          ],
          "bRange": [
            0.7575,
            1
          ],
          "calculation": true
        },
        {
          "metric": "Does a JSON schema raise the pass rate? Instructions vs schema mode (Right answer in any format (strict pass or format miss))",
          "aValue": 1,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "100% (12/12)",
          "bDisplay": "100% (12/12)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Code 76% to 100%; Codex CLI 76% to 100%), so this sample cannot separate them.",
          "studySlug": "json-schema-vs-instructions",
          "n": 12,
          "chartId": "structured-output-pass-rate",
          "aN": 12,
          "bN": 12,
          "aContext": "Claude Sonnet 5.5 · instructions",
          "bContext": "GPT-6.1 Sol · effort low · instructions",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.7575,
            1
          ],
          "bRange": [
            0.7575,
            1
          ],
          "calculation": true
        },
        {
          "metric": "What each call produced: strict pass, format miss, wrong values or error (Strict pass)",
          "aValue": 12,
          "bValue": 12,
          "unit": "count",
          "aDisplay": "12",
          "bDisplay": "12",
          "winner": "tie",
          "basis": "Same value. More or fewer is not better by itself for this metric.",
          "studySlug": "json-schema-vs-instructions",
          "n": 12,
          "chartId": "structured-output-outcomes",
          "aN": 12,
          "bN": 12,
          "aContext": "Claude Sonnet 5.5 · instructions",
          "bContext": "GPT-6.1 Sol · effort low · instructions"
        },
        {
          "metric": "What each call produced: strict pass, format miss, wrong values or error (Format miss)",
          "aValue": 0,
          "bValue": 0,
          "unit": "count",
          "aDisplay": "0",
          "bDisplay": "0",
          "winner": "tie",
          "basis": "Same value. More or fewer is not better by itself for this metric.",
          "studySlug": "json-schema-vs-instructions",
          "n": 12,
          "chartId": "structured-output-outcomes",
          "aN": 12,
          "bN": 12,
          "aContext": "Claude Sonnet 5.5 · instructions",
          "bContext": "GPT-6.1 Sol · effort low · instructions"
        },
        {
          "metric": "What each call produced: strict pass, format miss, wrong values or error (Wrong values)",
          "aValue": 0,
          "bValue": 0,
          "unit": "count",
          "aDisplay": "0",
          "bDisplay": "0",
          "winner": "tie",
          "basis": "Same value. More or fewer is not better by itself for this metric.",
          "studySlug": "json-schema-vs-instructions",
          "n": 12,
          "chartId": "structured-output-outcomes",
          "aN": 12,
          "bN": 12,
          "aContext": "Claude Sonnet 5.5 · instructions",
          "bContext": "GPT-6.1 Sol · effort low · instructions"
        },
        {
          "metric": "What each call produced: strict pass, format miss, wrong values or error (Error)",
          "aValue": 0,
          "bValue": 0,
          "unit": "count",
          "aDisplay": "0",
          "bDisplay": "0",
          "winner": "tie",
          "basis": "Same value. More or fewer is not better by itself for this metric.",
          "studySlug": "json-schema-vs-instructions",
          "n": 12,
          "chartId": "structured-output-outcomes",
          "aN": 12,
          "bN": 12,
          "aContext": "Claude Sonnet 5.5 · instructions",
          "bContext": "GPT-6.1 Sol · effort low · instructions"
        },
        {
          "metric": "Time per call, instructions vs schema mode",
          "aValue": 3.52,
          "bValue": 6.21,
          "unit": "seconds",
          "aDisplay": "3.52 s",
          "bDisplay": "6.21 s",
          "winner": "a",
          "basis": "The run ranges (fastest to slowest) do not overlap (Claude Code 2.67 s to 4.12 s; Codex CLI 4.20 s to 12.3 s). A range is not a confidence interval.",
          "studySlug": "json-schema-vs-instructions",
          "n": 12,
          "chartId": "structured-output-time",
          "aN": 12,
          "bN": 12,
          "aContext": "Claude Sonnet 5.5 · instructions",
          "bContext": "GPT-6.1 Sol · effort low · instructions",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            2.67,
            4.12
          ],
          "bRange": [
            4.2,
            12.27
          ]
        },
        {
          "metric": "Output and reasoning tokens per call, instructions vs schema mode (Median output tokens per call)",
          "aValue": 368,
          "bValue": 117,
          "unit": "tokens",
          "aDisplay": "368",
          "bDisplay": "117",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "json-schema-vs-instructions",
          "n": 12,
          "chartId": "structured-output-tokens",
          "aN": 12,
          "bN": 12,
          "aContext": "Claude Sonnet 5.5 · instructions",
          "bContext": "GPT-6.1 Sol · effort low · instructions"
        },
        {
          "metric": "Reasoning share of output tokens per call on hard tasks (calculation)",
          "aValue": 54.54,
          "bValue": 46.33,
          "unit": "percent",
          "aDisplay": "54.5%",
          "bDisplay": "46.3%",
          "winner": "unclear",
          "basis": "More or fewer percent is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-share",
          "aN": 24,
          "bN": 16,
          "aContext": "Claude Sonnet 5.5",
          "bContext": "GPT-6.1 Sol · effort medium",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            0,
            95.91
          ],
          "bRange": [
            11.42,
            86.85
          ],
          "calculation": true
        },
        {
          "metric": "List-price cost per call: reasoning, remaining output and input (calculation) (Reasoning (output tokens))",
          "aValue": 0.006665,
          "bValue": 0.002273,
          "unit": "usd",
          "aDisplay": "$0.0067",
          "bDisplay": "$0.0023",
          "winner": "unclear",
          "basis": "More or fewer usd is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-cost-per-call",
          "aN": 24,
          "bN": 16,
          "aContext": "Claude Sonnet 5.5",
          "bContext": "GPT-6.1 Sol · effort medium",
          "calculation": true
        },
        {
          "metric": "List-price cost per call: reasoning, remaining output and input (calculation) (Remaining output (visible-answer estimate))",
          "aValue": 0.003672,
          "bValue": 0.003025,
          "unit": "usd",
          "aDisplay": "$0.0037",
          "bDisplay": "$0.0030",
          "winner": "unclear",
          "basis": "More or fewer usd is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-cost-per-call",
          "aN": 24,
          "bN": 16,
          "aContext": "Claude Sonnet 5.5",
          "bContext": "GPT-6.1 Sol · effort medium",
          "calculation": true
        },
        {
          "metric": "List-price cost per call: reasoning, remaining output and input (calculation) (Input (prompt, cache priced))",
          "aValue": 0.004012,
          "bValue": 0.020339,
          "unit": "usd",
          "aDisplay": "$0.0040",
          "bDisplay": "$0.020",
          "winner": "unclear",
          "basis": "More or fewer usd is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-cost-per-call",
          "aN": 24,
          "bN": 16,
          "aContext": "Claude Sonnet 5.5",
          "bContext": "GPT-6.1 Sol · effort medium",
          "calculation": true
        },
        {
          "metric": "Reasoning cost per strict pass by effort, with the total (calculation) (Reasoning cost per strict pass)",
          "aValue": 0.005946,
          "bValue": 0.002273,
          "unit": "usd",
          "aDisplay": "$0.0059",
          "bDisplay": "$0.0023",
          "winner": "unclear",
          "basis": "More or fewer usd is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "n": 16,
          "chartId": "thinking-bill-by-effort",
          "aN": 16,
          "bN": 16,
          "aContext": "Claude Sonnet 5.5 · effort medium",
          "bContext": "GPT-6.1 Sol · effort medium",
          "calculation": true
        },
        {
          "metric": "Reasoning cost per strict pass by effort, with the total (calculation) (Total cost per strict pass)",
          "aValue": 0.01352,
          "bValue": 0.025637,
          "unit": "usd",
          "aDisplay": "$0.014",
          "bDisplay": "$0.026",
          "winner": "unclear",
          "basis": "More or fewer usd is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "n": 16,
          "chartId": "thinking-bill-by-effort",
          "aN": 16,
          "bN": 16,
          "aContext": "Claude Sonnet 5.5 · effort medium",
          "bContext": "GPT-6.1 Sol · effort medium",
          "calculation": true
        },
        {
          "metric": "Reasoning share on short tasks vs hard tasks (calculation) (Eight hard tasks)",
          "aValue": 54.54,
          "bValue": 46.33,
          "unit": "percent",
          "aDisplay": "54.5%",
          "bDisplay": "46.3%",
          "winner": "unclear",
          "basis": "More or fewer percent is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-short-vs-hard",
          "aN": 24,
          "bN": 16,
          "aContext": "Claude Sonnet 5.5",
          "bContext": "GPT-6.1 Sol · effort medium",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            0,
            95.91
          ],
          "bRange": [
            11.42,
            86.85
          ],
          "calculation": true
        },
        {
          "metric": "Reasoning share on short tasks vs hard tasks (calculation) (Five short tasks)",
          "aValue": 0,
          "bValue": 41.05,
          "unit": "percent",
          "aDisplay": "0%",
          "bDisplay": "41%",
          "winner": "unclear",
          "basis": "More or fewer percent is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "n": 15,
          "chartId": "thinking-bill-short-vs-hard",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Sonnet 5.5",
          "bContext": "GPT-6.1 Sol · effort medium",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            0,
            72.75
          ],
          "bRange": [
            0,
            71.43
          ],
          "calculation": true
        },
        {
          "metric": "Time to first text: a 250-line answer, six models",
          "aValue": 1.96,
          "bValue": 3.52,
          "unit": "seconds",
          "aDisplay": "1.96 s",
          "bDisplay": "3.52 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Code 0.88 s to 4.09 s; Codex CLI 2.75 s to 4.42 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "llm-speed-anatomy",
          "n": 4,
          "chartId": "speed-anatomy-first-text",
          "aN": 4,
          "bN": 4,
          "aContext": "Claude Sonnet 5.5",
          "bContext": "GPT-6.1 Sol · effort low",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            0.88,
            4.09
          ],
          "bRange": [
            2.75,
            4.42
          ]
        },
        {
          "metric": "Output speed after the first text: visible tokens per second (calculation)",
          "aValue": 231.7,
          "bValue": 79.6,
          "unit": "tokens",
          "aDisplay": "232",
          "bDisplay": "80",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "llm-speed-anatomy",
          "n": 4,
          "chartId": "speed-anatomy-output-speed",
          "aN": 4,
          "bN": 4,
          "aContext": "Claude Sonnet 5.5",
          "bContext": "GPT-6.1 Sol · effort low",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            230.3,
            233
          ],
          "bRange": [
            71.6,
            80.5
          ],
          "calculation": true
        },
        {
          "metric": "Output speed in characters per second after the first text (calculation)",
          "aValue": 517,
          "bValue": 323,
          "unit": "count",
          "aDisplay": "517",
          "bDisplay": "323",
          "winner": "unclear",
          "basis": "More or fewer count is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "llm-speed-anatomy",
          "n": 4,
          "chartId": "speed-anatomy-chars-per-second",
          "aN": 4,
          "bN": 4,
          "aContext": "Claude Sonnet 5.5",
          "bContext": "GPT-6.1 Sol · effort low",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            513,
            519
          ],
          "bRange": [
            291,
            327
          ],
          "calculation": true
        },
        {
          "metric": "Time to first text as the prompt grows: 1k",
          "aValue": 1.45,
          "bValue": 3.36,
          "unit": "seconds",
          "aDisplay": "1.45 s",
          "bDisplay": "3.36 s",
          "winner": "unclear",
          "basis": "Only 3 runs per side; the run ranges (fastest to slowest) do not overlap (Claude Code 1.23 s to 1.72 s; Codex CLI 3.36 s to 4.75 s), but 3 runs cannot show a reliable difference. A range is not a confidence interval.",
          "studySlug": "llm-speed-anatomy",
          "n": 3,
          "chartId": "speed-anatomy-prompt-size",
          "aN": 3,
          "bN": 3,
          "aContext": "Claude Sonnet 5.5",
          "bContext": "GPT-6.1 Sol · effort low",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            1.23,
            1.72
          ],
          "bRange": [
            3.36,
            4.75
          ],
          "calculation": true
        },
        {
          "metric": "Time to first text as the prompt grows: 16k",
          "aValue": 1.78,
          "bValue": 4.02,
          "unit": "seconds",
          "aDisplay": "1.78 s",
          "bDisplay": "4.02 s",
          "winner": "unclear",
          "basis": "Only 3 runs per side; the run ranges (fastest to slowest) do not overlap (Claude Code 1.64 s to 2.11 s; Codex CLI 3.30 s to 4.28 s), but 3 runs cannot show a reliable difference. A range is not a confidence interval.",
          "studySlug": "llm-speed-anatomy",
          "n": 3,
          "chartId": "speed-anatomy-prompt-size",
          "aN": 3,
          "bN": 3,
          "aContext": "Claude Sonnet 5.5",
          "bContext": "GPT-6.1 Sol · effort low",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            1.64,
            2.11
          ],
          "bRange": [
            3.3,
            4.28
          ],
          "calculation": true
        },
        {
          "metric": "Time to first text as the prompt grows: 64k",
          "aValue": 3.07,
          "bValue": 3.93,
          "unit": "seconds",
          "aDisplay": "3.07 s",
          "bDisplay": "3.93 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Code 1.38 s to 3.61 s; Codex CLI 3.42 s to 4.38 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "llm-speed-anatomy",
          "n": 3,
          "chartId": "speed-anatomy-prompt-size",
          "aN": 3,
          "bN": 3,
          "aContext": "Claude Sonnet 5.5",
          "bContext": "GPT-6.1 Sol · effort low",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            1.38,
            3.61
          ],
          "bRange": [
            3.42,
            4.38
          ],
          "calculation": true
        },
        {
          "metric": "Total time per call by prompt size (1k prompt)",
          "aValue": 1.78,
          "bValue": 3.43,
          "unit": "seconds",
          "aDisplay": "1.78 s",
          "bDisplay": "3.43 s",
          "winner": "unclear",
          "basis": "Only 3 runs per side; the run ranges (fastest to slowest) do not overlap (Claude Code 1.57 s to 2.12 s; Codex CLI 3.43 s to 4.92 s), but 3 runs cannot show a reliable difference. A range is not a confidence interval.",
          "studySlug": "llm-speed-anatomy",
          "n": 3,
          "chartId": "speed-anatomy-total-by-size",
          "aN": 3,
          "bN": 3,
          "aContext": "Claude Sonnet 5.5",
          "bContext": "GPT-6.1 Sol · effort low",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            1.57,
            2.12
          ],
          "bRange": [
            3.43,
            4.92
          ]
        },
        {
          "metric": "Total time per call by prompt size (16k prompt)",
          "aValue": 2.1,
          "bValue": 4.14,
          "unit": "seconds",
          "aDisplay": "2.10 s",
          "bDisplay": "4.14 s",
          "winner": "unclear",
          "basis": "Only 3 runs per side; the run ranges (fastest to slowest) do not overlap (Claude Code 1.98 s to 2.48 s; Codex CLI 3.96 s to 4.68 s), but 3 runs cannot show a reliable difference. A range is not a confidence interval.",
          "studySlug": "llm-speed-anatomy",
          "n": 3,
          "chartId": "speed-anatomy-total-by-size",
          "aN": 3,
          "bN": 3,
          "aContext": "Claude Sonnet 5.5",
          "bContext": "GPT-6.1 Sol · effort low",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            1.98,
            2.48
          ],
          "bRange": [
            3.96,
            4.68
          ]
        },
        {
          "metric": "Total time per call by prompt size (64k prompt)",
          "aValue": 3.44,
          "bValue": 3.96,
          "unit": "seconds",
          "aDisplay": "3.44 s",
          "bDisplay": "3.96 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Code 1.74 s to 4.38 s; Codex CLI 3.47 s to 4.44 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "llm-speed-anatomy",
          "n": 3,
          "chartId": "speed-anatomy-total-by-size",
          "aN": 3,
          "bN": 3,
          "aContext": "Claude Sonnet 5.5",
          "bContext": "GPT-6.1 Sol · effort low",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            1.74,
            4.38
          ],
          "bRange": [
            3.47,
            4.44
          ]
        },
        {
          "metric": "Exact lookup answers at the 1k, 16k and 64k prompt-size targets",
          "aValue": 1,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "100% (9/9)",
          "bDisplay": "100% (9/9)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Code 70% to 100%; Codex CLI 70% to 100%), so this sample cannot separate them.",
          "studySlug": "llm-speed-anatomy",
          "n": 9,
          "chartId": "speed-anatomy-lookup-correct",
          "aN": 9,
          "bN": 9,
          "aContext": "Claude Sonnet 5.5",
          "bContext": "GPT-6.1 Sol · effort low",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.7009,
            1
          ],
          "bRange": [
            0.7009,
            1
          ]
        },
        {
          "metric": "Pass rate on 4 harder tasks (Strict pass)",
          "aValue": 0.375,
          "bValue": 0.6875,
          "unit": "rate",
          "aDisplay": "38% (6/16)",
          "bDisplay": "69% (11/16)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Code 18% to 61%; Codex CLI 44% to 86%), so this sample cannot separate them.",
          "studySlug": "harder-tasks-head-to-head",
          "n": 16,
          "chartId": "harder-h2h-pass-rate",
          "aN": 16,
          "bN": 16,
          "aContext": "Claude Sonnet 5.5",
          "bContext": "GPT-6.1 Sol · effort medium",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.1848,
            0.6136
          ],
          "bRange": [
            0.444,
            0.8584
          ]
        },
        {
          "metric": "Pass rate on 4 harder tasks (Lenient (format misses counted))",
          "aValue": 0.375,
          "bValue": 0.6875,
          "unit": "rate",
          "aDisplay": "38% (6/16)",
          "bDisplay": "69% (11/16)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Code 18% to 61%; Codex CLI 44% to 86%), so this sample cannot separate them.",
          "studySlug": "harder-tasks-head-to-head",
          "n": 16,
          "chartId": "harder-h2h-pass-rate",
          "aN": 16,
          "bN": 16,
          "aContext": "Claude Sonnet 5.5",
          "bContext": "GPT-6.1 Sol · effort medium",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.1848,
            0.6136
          ],
          "bRange": [
            0.444,
            0.8584
          ]
        },
        {
          "metric": "Calls that tried a tool although tools were off",
          "aValue": 0.3125,
          "bValue": 0,
          "unit": "rate",
          "aDisplay": "31% (5/16)",
          "bDisplay": "0% (0/16)",
          "winner": "unclear",
          "basis": "More or fewer rate is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "harder-tasks-head-to-head",
          "n": 16,
          "chartId": "harder-h2h-tool-attempts",
          "aN": 16,
          "bN": 16,
          "aContext": "Claude Sonnet 5.5",
          "bContext": "GPT-6.1 Sol · effort medium",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.1416,
            0.556
          ],
          "bRange": [
            0,
            0.1936
          ]
        },
        {
          "metric": "Strict pass rate by task: 10x10 nonogram",
          "aValue": 1,
          "bValue": 0.75,
          "unit": "rate",
          "aDisplay": "100% (4/4)",
          "bDisplay": "75% (3/4)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Code 51% to 100%; Codex CLI 30% to 95%), so this sample cannot separate them.",
          "studySlug": "harder-tasks-head-to-head",
          "n": 4,
          "chartId": "harder-h2h-pass-by-task",
          "aN": 4,
          "bN": 4,
          "aContext": "Claude Sonnet 5.5",
          "bContext": "GPT-6.1 Sol · effort medium",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.5101,
            1
          ],
          "bRange": [
            0.3006,
            0.9544
          ]
        },
        {
          "metric": "Strict pass rate by task: Sudoku, 22 givens",
          "aValue": 0,
          "bValue": 0.25,
          "unit": "rate",
          "aDisplay": "0% (0/4)",
          "bDisplay": "25% (1/4)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Code 0% to 49%; Codex CLI 5% to 70%), so this sample cannot separate them.",
          "studySlug": "harder-tasks-head-to-head",
          "n": 4,
          "chartId": "harder-h2h-pass-by-task",
          "aN": 4,
          "bN": 4,
          "aContext": "Claude Sonnet 5.5",
          "bContext": "GPT-6.1 Sol · effort medium",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0,
            0.4899
          ],
          "bRange": [
            0.0456,
            0.6994
          ]
        },
        {
          "metric": "Strict pass rate by task: 6x6 Skyscrapers",
          "aValue": 0,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "0% (0/4)",
          "bDisplay": "100% (4/4)",
          "winner": "b",
          "basis": "The 95% intervals do not overlap (Claude Code 0% to 49%; Codex CLI 51% to 100%).",
          "studySlug": "harder-tasks-head-to-head",
          "n": 4,
          "chartId": "harder-h2h-pass-by-task",
          "aN": 4,
          "bN": 4,
          "aContext": "Claude Sonnet 5.5",
          "bContext": "GPT-6.1 Sol · effort medium",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0,
            0.4899
          ],
          "bRange": [
            0.5101,
            1
          ]
        },
        {
          "metric": "Strict pass rate by task: Seeded shuffle output",
          "aValue": 0.5,
          "bValue": 0.75,
          "unit": "rate",
          "aDisplay": "50% (2/4)",
          "bDisplay": "75% (3/4)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Code 15% to 85%; Codex CLI 30% to 95%), so this sample cannot separate them.",
          "studySlug": "harder-tasks-head-to-head",
          "n": 4,
          "chartId": "harder-h2h-pass-by-task",
          "aN": 4,
          "bN": 4,
          "aContext": "Claude Sonnet 5.5",
          "bContext": "GPT-6.1 Sol · effort medium",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.15,
            0.85
          ],
          "bRange": [
            0.3006,
            0.9544
          ]
        },
        {
          "metric": "Total time per call on harder tasks",
          "aValue": 70.43,
          "bValue": 120.24,
          "unit": "seconds",
          "aDisplay": "70.4 s",
          "bDisplay": "120.2 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Code 4.32 s to 210.1 s; Codex CLI 46.2 s to 273.5 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-total-latency",
          "aN": 12,
          "bN": 13,
          "aContext": "Claude Sonnet 5.5",
          "bContext": "GPT-6.1 Sol · effort medium",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            4.32,
            210.08
          ],
          "bRange": [
            46.24,
            273.46
          ]
        },
        {
          "metric": "Output tokens per call on harder tasks (Output tokens)",
          "aValue": 9287,
          "bValue": 4994,
          "unit": "tokens",
          "aDisplay": "9,287",
          "bDisplay": "4,994",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-output-tokens",
          "aN": 12,
          "bN": 13,
          "aContext": "Claude Sonnet 5.5",
          "bContext": "GPT-6.1 Sol · effort medium",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            407,
            27921
          ],
          "bRange": [
            2099,
            13413
          ]
        },
        {
          "metric": "List-price cost per strict pass on harder tasks (calculation)",
          "aValue": 0.23843,
          "bValue": 0.08293,
          "unit": "usd",
          "aDisplay": "$0.24",
          "bDisplay": "$0.083",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.24 vs $0.083, 2.9x) is not tested against run-to-run variation.",
          "studySlug": "harder-tasks-head-to-head",
          "n": 16,
          "chartId": "harder-h2h-cost-per-pass",
          "aN": 16,
          "bN": 16,
          "aContext": "Claude Sonnet 5.5",
          "bContext": "GPT-6.1 Sol · effort medium",
          "calculation": true
        }
      ]
    },
    {
      "slug": "claude-sonnet-5-5-vs-gpt-6-1-sol-codex-cli",
      "a": "claude-sonnet-5-5",
      "b": "gpt-6-1-sol-codex-cli",
      "title": "Claude Sonnet 5.5 vs GPT-6.1 Sol (Codex CLI)",
      "seoTitle": "Claude Sonnet 5.5 vs GPT-6.1 Sol (Codex CLI): benchmarks",
      "description": "Claude Sonnet 5.5 vs GPT-6.1 Sol (Codex CLI): 49 measured metrics from 9 studies (Pass rate on five validated tasks; more), with sample sizes and intervals.",
      "verdict": "Claude Sonnet 5.5 and GPT-6.1 Sol (Codex CLI) share 49 measured metrics and 21 list-price calculations from 10 studies. Claude Sonnet 5.5 leads on 4 rows: Time per coding session, 23.1 s vs 113.4 s; Same prompt, 10 times: time per call (Exact number), 6.89 s vs 13.4 s; Same prompt, 10 times: time per call (Code fix), 2.67 s vs 11.3 s; and 1 more. GPT-6.1 Sol (Codex CLI) leads on 1 row: Strict pass rate by task: 6x6 Skyscrapers, 100% (4/4) vs 0% (0/4). On those rows the 95% intervals and run ranges do not overlap; only a 95% interval is a confidence interval. The other rows are 22 ties and 43 unclear; each row says why. Calculation rows are derived from list prices and recorded counts; they are not bills or runs. Every row ran the two sides through different routes (for example Claude Code vs Codex CLI), so they compare route + model pairs, not models alone; the contexts name the route. Some rows rest on small samples (n = 3 at the smallest).",
      "rows": [
        {
          "metric": "Pass rate on five validated tasks",
          "aValue": 0.8,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "80% (12/15)",
          "bDisplay": "100% (15/15)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Sonnet 5.5 55% to 93%; GPT-6.1 Sol (Codex CLI) 80% to 100%), so this sample cannot separate them.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-pass-rate",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code · five short validated tasks",
          "bContext": "Codex CLI · effort medium · five short validated tasks",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.5481,
            0.9295
          ],
          "bRange": [
            0.7961,
            1
          ]
        },
        {
          "metric": "Total time per call",
          "aValue": 2.31,
          "bValue": 5.65,
          "unit": "seconds",
          "aDisplay": "2.31 s",
          "bDisplay": "5.65 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Sonnet 5.5 2.17 s to 7.73 s; GPT-6.1 Sol (Codex CLI) 4.10 s to 25.5 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-total-latency",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code · five short validated tasks",
          "bContext": "Codex CLI · effort medium · five short validated tasks",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            2.17,
            7.73
          ],
          "bRange": [
            4.1,
            25.46
          ]
        },
        {
          "metric": "Time to first useful output",
          "aValue": 1.56,
          "bValue": 5.05,
          "unit": "seconds",
          "aDisplay": "1.56 s",
          "bDisplay": "5.05 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Sonnet 5.5 0.99 s to 6.39 s; GPT-6.1 Sol (Codex CLI) 3.36 s to 17.8 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-first-useful-latency",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code · five short validated tasks",
          "bContext": "Codex CLI · effort medium · five short validated tasks",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            0.99,
            6.39
          ],
          "bRange": [
            3.36,
            17.82
          ]
        },
        {
          "metric": "Input tokens per call: what the CLI sends (Cache read)",
          "aValue": 1401,
          "bValue": 5180,
          "unit": "tokens",
          "aDisplay": "1,401",
          "bDisplay": "5,180",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-input-tokens",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code · five short validated tasks",
          "bContext": "Codex CLI · effort medium · five short validated tasks"
        },
        {
          "metric": "Input tokens per call: what the CLI sends (Other input)",
          "aValue": 685,
          "bValue": 6943,
          "unit": "tokens",
          "aDisplay": "685",
          "bDisplay": "6,943",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-input-tokens",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code · five short validated tasks",
          "bContext": "Codex CLI · effort medium · five short validated tasks"
        },
        {
          "metric": "Output tokens per call (Output tokens)",
          "aValue": 107,
          "bValue": 42,
          "unit": "tokens",
          "aDisplay": "107",
          "bDisplay": "42",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-output-tokens",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code · five short validated tasks",
          "bContext": "Codex CLI · effort medium · five short validated tasks"
        },
        {
          "metric": "List-price cost per call (calculation)",
          "aValue": 0.0036,
          "bValue": 0.01018,
          "unit": "usd",
          "aDisplay": "$0.0036",
          "bDisplay": "$0.010",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Sonnet 5.5 $0.0034 to $0.010; GPT-6.1 Sol (Codex CLI) $0.0054 to $0.027); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-list-price-per-call",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code · five short validated tasks",
          "bContext": "Codex CLI · effort medium · five short validated tasks",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            0.00342,
            0.01021
          ],
          "bRange": [
            0.0054,
            0.02686
          ],
          "calculation": true
        },
        {
          "metric": "List-price cost per passing answer (calculation)",
          "aValue": 0.00624,
          "bValue": 0.01564,
          "unit": "usd",
          "aDisplay": "$0.0062",
          "bDisplay": "$0.016",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.0062 vs $0.016, 2.5x) is not tested against run-to-run variation.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-cost-per-pass",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code · five short validated tasks",
          "bContext": "Codex CLI · effort medium · five short validated tasks",
          "calculation": true
        },
        {
          "metric": "Pass rate on eight hard tasks (Strict pass)",
          "aValue": 1,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "100% (24/24)",
          "bDisplay": "100% (16/16)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Sonnet 5.5 86% to 100%; GPT-6.1 Sol (Codex CLI) 81% to 100%), so this sample cannot separate them.",
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-pass-rate",
          "aN": 24,
          "bN": 16,
          "aContext": "Claude Code · eight hard validated tasks",
          "bContext": "Codex CLI · effort medium · eight hard validated tasks",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.862,
            1
          ],
          "bRange": [
            0.8064,
            1
          ]
        },
        {
          "metric": "Pass rate on eight hard tasks (Lenient (format misses counted))",
          "aValue": 1,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "100% (24/24)",
          "bDisplay": "100% (16/16)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Sonnet 5.5 86% to 100%; GPT-6.1 Sol (Codex CLI) 81% to 100%), so this sample cannot separate them.",
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-pass-rate",
          "aN": 24,
          "bN": 16,
          "aContext": "Claude Code · eight hard validated tasks",
          "bContext": "Codex CLI · effort medium · eight hard validated tasks",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.862,
            1
          ],
          "bRange": [
            0.8064,
            1
          ]
        },
        {
          "metric": "Total time per call on hard tasks (separate batches)",
          "aValue": 7.75,
          "bValue": 13.11,
          "unit": "seconds",
          "aDisplay": "7.75 s",
          "bDisplay": "13.1 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Sonnet 5.5 2.26 s to 34.8 s; GPT-6.1 Sol (Codex CLI) 8.54 s to 61.6 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-total-latency",
          "aN": 24,
          "bN": 16,
          "aContext": "Claude Code · eight hard validated tasks",
          "bContext": "Codex CLI · effort medium · eight hard validated tasks",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            2.26,
            34.79
          ],
          "bRange": [
            8.54,
            61.6
          ]
        },
        {
          "metric": "Time to first useful output on hard tasks",
          "aValue": 5.95,
          "bValue": 10.23,
          "unit": "seconds",
          "aDisplay": "5.95 s",
          "bDisplay": "10.2 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Sonnet 5.5 0.86 s to 30.6 s; GPT-6.1 Sol (Codex CLI) 6.09 s to 40.4 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-first-useful-latency",
          "aN": 24,
          "bN": 16,
          "aContext": "Claude Code · eight hard validated tasks",
          "bContext": "Codex CLI · effort medium · eight hard validated tasks",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            0.86,
            30.57
          ],
          "bRange": [
            6.09,
            40.41
          ]
        },
        {
          "metric": "Output tokens per call on hard tasks (Output tokens)",
          "aValue": 1050,
          "bValue": 335,
          "unit": "tokens",
          "aDisplay": "1,050",
          "bDisplay": "335",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-output-tokens",
          "aN": 24,
          "bN": 16,
          "aContext": "Claude Code · eight hard validated tasks",
          "bContext": "Codex CLI · effort medium · eight hard validated tasks"
        },
        {
          "metric": "List-price cost per strict pass on hard tasks (calculation)",
          "aValue": 0.01435,
          "bValue": 0.02564,
          "unit": "usd",
          "aDisplay": "$0.014",
          "bDisplay": "$0.026",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.014 vs $0.026) is not tested against run-to-run variation.",
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-cost-per-pass",
          "aN": 24,
          "bN": 16,
          "aContext": "Claude Code · eight hard validated tasks",
          "bContext": "Codex CLI · effort medium · eight hard validated tasks",
          "calculation": true
        },
        {
          "metric": "Coding sessions that passed every hidden check",
          "aValue": 1,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "100% (12/12)",
          "bDisplay": "100% (12/12)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Sonnet 5.5 76% to 100%; GPT-6.1 Sol (Codex CLI) 76% to 100%), so this sample cannot separate them.",
          "studySlug": "coding-agents-head-to-head",
          "n": 12,
          "chartId": "coding-agents-pass-rate",
          "aN": 12,
          "bN": 12,
          "aContext": "Claude Code · six small repository tasks with hidden tests",
          "bContext": "Codex CLI · effort medium · tester’s AGENTS.md · six small repository tasks with hidden tests",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.7575,
            1
          ],
          "bRange": [
            0.7575,
            1
          ]
        },
        {
          "metric": "Time per coding session",
          "aValue": 23.1,
          "bValue": 113.4,
          "unit": "seconds",
          "aDisplay": "23.1 s",
          "bDisplay": "113.4 s",
          "winner": "a",
          "basis": "The run ranges (fastest to slowest) do not overlap (Claude Sonnet 5.5 18.7 s to 44.5 s; GPT-6.1 Sol (Codex CLI) 78.5 s to 221.9 s). A range is not a confidence interval.",
          "studySlug": "coding-agents-head-to-head",
          "n": 12,
          "chartId": "coding-agents-wall-time",
          "aN": 12,
          "bN": 12,
          "aContext": "Claude Code · six small repository tasks with hidden tests",
          "bContext": "Codex CLI · effort medium · tester’s AGENTS.md · six small repository tasks with hidden tests",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            18.7,
            44.5
          ],
          "bRange": [
            78.5,
            221.9
          ]
        },
        {
          "metric": "Tool calls per coding session",
          "aValue": 7.5,
          "bValue": 12.5,
          "unit": "calls",
          "aDisplay": "7.5",
          "bDisplay": "12.5",
          "winner": "unclear",
          "basis": "More or fewer calls is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "coding-agents-head-to-head",
          "n": 12,
          "chartId": "coding-agents-tool-calls",
          "aN": 12,
          "bN": 12,
          "aContext": "Claude Code · six small repository tasks with hidden tests",
          "bContext": "Codex CLI · effort medium · tester’s AGENTS.md · six small repository tasks with hidden tests",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            3,
            14
          ],
          "bRange": [
            8,
            18
          ]
        },
        {
          "metric": "List-price cost per passing coding session (calculation)",
          "aValue": 0.085,
          "bValue": 0.0978,
          "unit": "usd",
          "aDisplay": "$0.085",
          "bDisplay": "$0.098",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.085 vs $0.098) is not tested against run-to-run variation.",
          "studySlug": "coding-agents-head-to-head",
          "n": 12,
          "chartId": "coding-agents-cost-per-pass",
          "aN": 12,
          "bN": 12,
          "aContext": "Claude Code · six small repository tasks with hidden tests",
          "bContext": "Codex CLI · effort medium · tester’s AGENTS.md · six small repository tasks with hidden tests",
          "calculation": true
        },
        {
          "metric": "Strict pass rate by effort on eight hard tasks",
          "aValue": 1,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "100% (16/16)",
          "bDisplay": "100% (16/16)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Sonnet 5.5 81% to 100%; GPT-6.1 Sol (Codex CLI) 81% to 100%), so this sample cannot separate them.",
          "studySlug": "effort-ladder",
          "n": 16,
          "chartId": "effort-ladder-pass-rate",
          "aN": 16,
          "bN": 16,
          "aContext": "Claude Code · effort medium · eight hard validated tasks, effort ladder",
          "bContext": "Codex CLI · effort medium · eight hard validated tasks, effort ladder",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.8064,
            1
          ],
          "bRange": [
            0.8064,
            1
          ]
        },
        {
          "metric": "Total time per call by effort on hard tasks",
          "aValue": 7.63,
          "bValue": 13.11,
          "unit": "seconds",
          "aDisplay": "7.63 s",
          "bDisplay": "13.1 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Sonnet 5.5 2.71 s to 24.0 s; GPT-6.1 Sol (Codex CLI) 8.54 s to 61.6 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "effort-ladder",
          "n": 16,
          "chartId": "effort-ladder-total-latency",
          "aN": 16,
          "bN": 16,
          "aContext": "Claude Code · effort medium · eight hard validated tasks, effort ladder",
          "bContext": "Codex CLI · effort medium · eight hard validated tasks, effort ladder",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            2.71,
            24.01
          ],
          "bRange": [
            8.54,
            61.6
          ]
        },
        {
          "metric": "Output tokens per call by effort on hard tasks (Output tokens)",
          "aValue": 770,
          "bValue": 335,
          "unit": "tokens",
          "aDisplay": "770",
          "bDisplay": "335",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "effort-ladder",
          "n": 16,
          "chartId": "effort-ladder-output-tokens",
          "aN": 16,
          "bN": 16,
          "aContext": "Claude Code · effort medium · eight hard validated tasks, effort ladder",
          "bContext": "Codex CLI · effort medium · eight hard validated tasks, effort ladder"
        },
        {
          "metric": "List-price cost per strict pass by effort (calculation)",
          "aValue": 0.01352,
          "bValue": 0.02564,
          "unit": "usd",
          "aDisplay": "$0.014",
          "bDisplay": "$0.026",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.014 vs $0.026) is not tested against run-to-run variation.",
          "studySlug": "effort-ladder",
          "n": 16,
          "chartId": "effort-ladder-cost-per-pass",
          "aN": 16,
          "bN": 16,
          "aContext": "Claude Code · effort medium · eight hard validated tasks, effort ladder",
          "bContext": "Codex CLI · effort medium · eight hard validated tasks, effort ladder",
          "calculation": true
        },
        {
          "metric": "Same prompt, 10 times: strict pass rate (Exact number)",
          "aValue": 1,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "100% (10/10)",
          "bDisplay": "100% (10/10)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Sonnet 5.5 72% to 100%; GPT-6.1 Sol (Codex CLI) 72% to 100%), so this sample cannot separate them.",
          "studySlug": "caching-consistency",
          "n": 10,
          "chartId": "consistency-pass-rate",
          "aN": 10,
          "bN": 10,
          "aContext": "Claude Code · same prompt repeated 10 times",
          "bContext": "Codex CLI · effort medium · same prompt repeated 10 times",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.7225,
            1
          ],
          "bRange": [
            0.7225,
            1
          ]
        },
        {
          "metric": "Same prompt, 10 times: strict pass rate (JSON object)",
          "aValue": 1,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "100% (10/10)",
          "bDisplay": "100% (10/10)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Sonnet 5.5 72% to 100%; GPT-6.1 Sol (Codex CLI) 72% to 100%), so this sample cannot separate them.",
          "studySlug": "caching-consistency",
          "n": 10,
          "chartId": "consistency-pass-rate",
          "aN": 10,
          "bN": 10,
          "aContext": "Claude Code · same prompt repeated 10 times",
          "bContext": "Codex CLI · effort medium · same prompt repeated 10 times",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.7225,
            1
          ],
          "bRange": [
            0.7225,
            1
          ]
        },
        {
          "metric": "Same prompt, 10 times: strict pass rate (Code fix)",
          "aValue": 1,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "100% (10/10)",
          "bDisplay": "100% (10/10)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Sonnet 5.5 72% to 100%; GPT-6.1 Sol (Codex CLI) 72% to 100%), so this sample cannot separate them.",
          "studySlug": "caching-consistency",
          "n": 10,
          "chartId": "consistency-pass-rate",
          "aN": 10,
          "bN": 10,
          "aContext": "Claude Code · same prompt repeated 10 times",
          "bContext": "Codex CLI · effort medium · same prompt repeated 10 times",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.7225,
            1
          ],
          "bRange": [
            0.7225,
            1
          ]
        },
        {
          "metric": "Same prompt, 10 times: how many different answers (Exact number)",
          "aValue": 1,
          "bValue": 1,
          "unit": "count",
          "aDisplay": "1",
          "bDisplay": "1",
          "winner": "tie",
          "basis": "Same value. More or fewer is not better by itself for this metric.",
          "studySlug": "caching-consistency",
          "n": 10,
          "chartId": "consistency-distinct-answers",
          "aN": 10,
          "bN": 10,
          "aContext": "Claude Code · same prompt repeated 10 times",
          "bContext": "Codex CLI · effort medium · same prompt repeated 10 times"
        },
        {
          "metric": "Same prompt, 10 times: how many different answers (JSON object)",
          "aValue": 1,
          "bValue": 1,
          "unit": "count",
          "aDisplay": "1",
          "bDisplay": "1",
          "winner": "tie",
          "basis": "Same value. More or fewer is not better by itself for this metric.",
          "studySlug": "caching-consistency",
          "n": 10,
          "chartId": "consistency-distinct-answers",
          "aN": 10,
          "bN": 10,
          "aContext": "Claude Code · same prompt repeated 10 times",
          "bContext": "Codex CLI · effort medium · same prompt repeated 10 times"
        },
        {
          "metric": "Same prompt, 10 times: how many different answers (Code fix)",
          "aValue": 3,
          "bValue": 6,
          "unit": "count",
          "aDisplay": "3",
          "bDisplay": "6",
          "winner": "unclear",
          "basis": "More or fewer count is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "caching-consistency",
          "n": 10,
          "chartId": "consistency-distinct-answers",
          "aN": 10,
          "bN": 10,
          "aContext": "Claude Code · same prompt repeated 10 times",
          "bContext": "Codex CLI · effort medium · same prompt repeated 10 times"
        },
        {
          "metric": "Same prompt, 10 times: time per call (Exact number)",
          "aValue": 6.89,
          "bValue": 13.38,
          "unit": "seconds",
          "aDisplay": "6.89 s",
          "bDisplay": "13.4 s",
          "winner": "a",
          "basis": "The run ranges (fastest to slowest) do not overlap (Claude Sonnet 5.5 5.81 s to 7.81 s; GPT-6.1 Sol (Codex CLI) 12.3 s to 18.0 s). A range is not a confidence interval.",
          "studySlug": "caching-consistency",
          "n": 10,
          "chartId": "consistency-latency-spread",
          "aN": 10,
          "bN": 10,
          "aContext": "Claude Code · same prompt repeated 10 times",
          "bContext": "Codex CLI · effort medium · same prompt repeated 10 times",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            5.81,
            7.81
          ],
          "bRange": [
            12.29,
            17.97
          ]
        },
        {
          "metric": "Same prompt, 10 times: time per call (JSON object)",
          "aValue": 2.89,
          "bValue": 6.42,
          "unit": "seconds",
          "aDisplay": "2.89 s",
          "bDisplay": "6.42 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Sonnet 5.5 2.68 s to 5.30 s; GPT-6.1 Sol (Codex CLI) 5.25 s to 8.26 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "caching-consistency",
          "n": 10,
          "chartId": "consistency-latency-spread",
          "aN": 10,
          "bN": 10,
          "aContext": "Claude Code · same prompt repeated 10 times",
          "bContext": "Codex CLI · effort medium · same prompt repeated 10 times",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            2.68,
            5.3
          ],
          "bRange": [
            5.25,
            8.26
          ]
        },
        {
          "metric": "Same prompt, 10 times: time per call (Code fix)",
          "aValue": 2.67,
          "bValue": 11.29,
          "unit": "seconds",
          "aDisplay": "2.67 s",
          "bDisplay": "11.3 s",
          "winner": "a",
          "basis": "The run ranges (fastest to slowest) do not overlap (Claude Sonnet 5.5 2.32 s to 4.34 s; GPT-6.1 Sol (Codex CLI) 9.08 s to 14.8 s). A range is not a confidence interval.",
          "studySlug": "caching-consistency",
          "n": 10,
          "chartId": "consistency-latency-spread",
          "aN": 10,
          "bN": 10,
          "aContext": "Claude Code · same prompt repeated 10 times",
          "bContext": "Codex CLI · effort medium · same prompt repeated 10 times",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            2.32,
            4.34
          ],
          "bRange": [
            9.08,
            14.85
          ]
        },
        {
          "metric": "Repairing a scheduler: Claude Code vs Codex vs API (Total time)",
          "aValue": 15,
          "bValue": 61.16,
          "unit": "seconds",
          "aDisplay": "15.0 s",
          "bDisplay": "61.2 s",
          "winner": "unclear",
          "basis": "Only 3 runs per side; the run ranges (fastest to slowest) do not overlap (Claude Sonnet 5.5 13.9 s to 15.9 s; GPT-6.1 Sol (Codex CLI) 59.9 s to 69.5 s), but 3 runs cannot show a reliable difference. A range is not a confidence interval.",
          "studySlug": "cli-model-latency-tokens",
          "n": 3,
          "chartId": "scheduler-repair-claude-vs-codex",
          "aN": 3,
          "bN": 3,
          "aContext": "Claude Code CLI · effort medium · scheduler repair, 296 checks, 3 runs",
          "bContext": "Codex CLI · effort medium · scheduler repair, 296 checks, 3 runs",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            13.89,
            15.89
          ],
          "bRange": [
            59.9,
            69.51
          ]
        },
        {
          "metric": "Repairing a scheduler: Claude Code vs Codex vs API (First useful output)",
          "aValue": 7.55,
          "bValue": 15.56,
          "unit": "seconds",
          "aDisplay": "7.55 s",
          "bDisplay": "15.6 s",
          "winner": "unclear",
          "basis": "Only 3 runs per side; the run ranges (fastest to slowest) do not overlap (Claude Sonnet 5.5 6.77 s to 7.63 s; GPT-6.1 Sol (Codex CLI) 13.7 s to 23.0 s), but 3 runs cannot show a reliable difference. A range is not a confidence interval.",
          "studySlug": "cli-model-latency-tokens",
          "n": 3,
          "chartId": "scheduler-repair-claude-vs-codex",
          "aN": 3,
          "bN": 3,
          "aContext": "Claude Code CLI · effort medium · scheduler repair, 296 checks, 3 runs",
          "bContext": "Codex CLI · effort medium · scheduler repair, 296 checks, 3 runs",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            6.77,
            7.63
          ],
          "bRange": [
            13.65,
            23.04
          ]
        },
        {
          "metric": "Output tokens to repair the scheduler (Output tokens)",
          "aValue": 2227,
          "bValue": 1181,
          "unit": "tokens",
          "aDisplay": "2,227",
          "bDisplay": "1,181",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "cli-model-latency-tokens",
          "n": 3,
          "chartId": "scheduler-repair-output-tokens",
          "aN": 3,
          "bN": 3,
          "aContext": "Claude Code CLI · effort medium · scheduler repair, 296 checks, 3 runs",
          "bContext": "Codex CLI · effort medium · scheduler repair, 296 checks, 3 runs"
        },
        {
          "metric": "Does a JSON schema raise the pass rate? Instructions vs schema mode (Strict pass: the whole reply is the right JSON)",
          "aValue": 1,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "100% (12/12)",
          "bDisplay": "100% (12/12)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Sonnet 5.5 76% to 100%; GPT-6.1 Sol (Codex CLI) 76% to 100%), so this sample cannot separate them.",
          "studySlug": "json-schema-vs-instructions",
          "n": 12,
          "chartId": "structured-output-pass-rate",
          "aN": 12,
          "bN": 12,
          "aContext": "Claude Code · instructions",
          "bContext": "Codex CLI · effort low · instructions",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.7575,
            1
          ],
          "bRange": [
            0.7575,
            1
          ],
          "calculation": true
        },
        {
          "metric": "Does a JSON schema raise the pass rate? Instructions vs schema mode (Right answer in any format (strict pass or format miss))",
          "aValue": 1,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "100% (12/12)",
          "bDisplay": "100% (12/12)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Sonnet 5.5 76% to 100%; GPT-6.1 Sol (Codex CLI) 76% to 100%), so this sample cannot separate them.",
          "studySlug": "json-schema-vs-instructions",
          "n": 12,
          "chartId": "structured-output-pass-rate",
          "aN": 12,
          "bN": 12,
          "aContext": "Claude Code · instructions",
          "bContext": "Codex CLI · effort low · instructions",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.7575,
            1
          ],
          "bRange": [
            0.7575,
            1
          ],
          "calculation": true
        },
        {
          "metric": "What each call produced: strict pass, format miss, wrong values or error (Strict pass)",
          "aValue": 12,
          "bValue": 12,
          "unit": "count",
          "aDisplay": "12",
          "bDisplay": "12",
          "winner": "tie",
          "basis": "Same value. More or fewer is not better by itself for this metric.",
          "studySlug": "json-schema-vs-instructions",
          "n": 12,
          "chartId": "structured-output-outcomes",
          "aN": 12,
          "bN": 12,
          "aContext": "Claude Code · instructions",
          "bContext": "Codex CLI · effort low · instructions"
        },
        {
          "metric": "What each call produced: strict pass, format miss, wrong values or error (Format miss)",
          "aValue": 0,
          "bValue": 0,
          "unit": "count",
          "aDisplay": "0",
          "bDisplay": "0",
          "winner": "tie",
          "basis": "Same value. More or fewer is not better by itself for this metric.",
          "studySlug": "json-schema-vs-instructions",
          "n": 12,
          "chartId": "structured-output-outcomes",
          "aN": 12,
          "bN": 12,
          "aContext": "Claude Code · instructions",
          "bContext": "Codex CLI · effort low · instructions"
        },
        {
          "metric": "What each call produced: strict pass, format miss, wrong values or error (Wrong values)",
          "aValue": 0,
          "bValue": 0,
          "unit": "count",
          "aDisplay": "0",
          "bDisplay": "0",
          "winner": "tie",
          "basis": "Same value. More or fewer is not better by itself for this metric.",
          "studySlug": "json-schema-vs-instructions",
          "n": 12,
          "chartId": "structured-output-outcomes",
          "aN": 12,
          "bN": 12,
          "aContext": "Claude Code · instructions",
          "bContext": "Codex CLI · effort low · instructions"
        },
        {
          "metric": "What each call produced: strict pass, format miss, wrong values or error (Error)",
          "aValue": 0,
          "bValue": 0,
          "unit": "count",
          "aDisplay": "0",
          "bDisplay": "0",
          "winner": "tie",
          "basis": "Same value. More or fewer is not better by itself for this metric.",
          "studySlug": "json-schema-vs-instructions",
          "n": 12,
          "chartId": "structured-output-outcomes",
          "aN": 12,
          "bN": 12,
          "aContext": "Claude Code · instructions",
          "bContext": "Codex CLI · effort low · instructions"
        },
        {
          "metric": "Time per call, instructions vs schema mode",
          "aValue": 3.52,
          "bValue": 6.21,
          "unit": "seconds",
          "aDisplay": "3.52 s",
          "bDisplay": "6.21 s",
          "winner": "a",
          "basis": "The run ranges (fastest to slowest) do not overlap (Claude Sonnet 5.5 2.67 s to 4.12 s; GPT-6.1 Sol (Codex CLI) 4.20 s to 12.3 s). A range is not a confidence interval.",
          "studySlug": "json-schema-vs-instructions",
          "n": 12,
          "chartId": "structured-output-time",
          "aN": 12,
          "bN": 12,
          "aContext": "Claude Code · instructions",
          "bContext": "Codex CLI · effort low · instructions",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            2.67,
            4.12
          ],
          "bRange": [
            4.2,
            12.27
          ]
        },
        {
          "metric": "Output and reasoning tokens per call, instructions vs schema mode (Median output tokens per call)",
          "aValue": 368,
          "bValue": 117,
          "unit": "tokens",
          "aDisplay": "368",
          "bDisplay": "117",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "json-schema-vs-instructions",
          "n": 12,
          "chartId": "structured-output-tokens",
          "aN": 12,
          "bN": 12,
          "aContext": "Claude Code · instructions",
          "bContext": "Codex CLI · effort low · instructions"
        },
        {
          "metric": "Reasoning share of output tokens per call on hard tasks (calculation)",
          "aValue": 54.54,
          "bValue": 46.33,
          "unit": "percent",
          "aDisplay": "54.5%",
          "bDisplay": "46.3%",
          "winner": "unclear",
          "basis": "More or fewer percent is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-share",
          "aN": 24,
          "bN": 16,
          "aContext": "Claude Code",
          "bContext": "Codex CLI · effort medium",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            0,
            95.91
          ],
          "bRange": [
            11.42,
            86.85
          ],
          "calculation": true
        },
        {
          "metric": "List-price cost per call: reasoning, remaining output and input (calculation) (Reasoning (output tokens))",
          "aValue": 0.006665,
          "bValue": 0.002273,
          "unit": "usd",
          "aDisplay": "$0.0067",
          "bDisplay": "$0.0023",
          "winner": "unclear",
          "basis": "More or fewer usd is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-cost-per-call",
          "aN": 24,
          "bN": 16,
          "aContext": "Claude Code",
          "bContext": "Codex CLI · effort medium",
          "calculation": true
        },
        {
          "metric": "List-price cost per call: reasoning, remaining output and input (calculation) (Remaining output (visible-answer estimate))",
          "aValue": 0.003672,
          "bValue": 0.003025,
          "unit": "usd",
          "aDisplay": "$0.0037",
          "bDisplay": "$0.0030",
          "winner": "unclear",
          "basis": "More or fewer usd is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-cost-per-call",
          "aN": 24,
          "bN": 16,
          "aContext": "Claude Code",
          "bContext": "Codex CLI · effort medium",
          "calculation": true
        },
        {
          "metric": "List-price cost per call: reasoning, remaining output and input (calculation) (Input (prompt, cache priced))",
          "aValue": 0.004012,
          "bValue": 0.020339,
          "unit": "usd",
          "aDisplay": "$0.0040",
          "bDisplay": "$0.020",
          "winner": "unclear",
          "basis": "More or fewer usd is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-cost-per-call",
          "aN": 24,
          "bN": 16,
          "aContext": "Claude Code",
          "bContext": "Codex CLI · effort medium",
          "calculation": true
        },
        {
          "metric": "Reasoning cost per strict pass by effort, with the total (calculation) (Reasoning cost per strict pass)",
          "aValue": 0.005946,
          "bValue": 0.002273,
          "unit": "usd",
          "aDisplay": "$0.0059",
          "bDisplay": "$0.0023",
          "winner": "unclear",
          "basis": "More or fewer usd is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "n": 16,
          "chartId": "thinking-bill-by-effort",
          "aN": 16,
          "bN": 16,
          "aContext": "Claude Code · effort medium",
          "bContext": "Codex CLI · effort medium",
          "calculation": true
        },
        {
          "metric": "Reasoning cost per strict pass by effort, with the total (calculation) (Total cost per strict pass)",
          "aValue": 0.01352,
          "bValue": 0.025637,
          "unit": "usd",
          "aDisplay": "$0.014",
          "bDisplay": "$0.026",
          "winner": "unclear",
          "basis": "More or fewer usd is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "n": 16,
          "chartId": "thinking-bill-by-effort",
          "aN": 16,
          "bN": 16,
          "aContext": "Claude Code · effort medium",
          "bContext": "Codex CLI · effort medium",
          "calculation": true
        },
        {
          "metric": "Reasoning share on short tasks vs hard tasks (calculation) (Eight hard tasks)",
          "aValue": 54.54,
          "bValue": 46.33,
          "unit": "percent",
          "aDisplay": "54.5%",
          "bDisplay": "46.3%",
          "winner": "unclear",
          "basis": "More or fewer percent is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-short-vs-hard",
          "aN": 24,
          "bN": 16,
          "aContext": "Claude Code",
          "bContext": "Codex CLI · effort medium",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            0,
            95.91
          ],
          "bRange": [
            11.42,
            86.85
          ],
          "calculation": true
        },
        {
          "metric": "Reasoning share on short tasks vs hard tasks (calculation) (Five short tasks)",
          "aValue": 0,
          "bValue": 41.05,
          "unit": "percent",
          "aDisplay": "0%",
          "bDisplay": "41%",
          "winner": "unclear",
          "basis": "More or fewer percent is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "n": 15,
          "chartId": "thinking-bill-short-vs-hard",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code",
          "bContext": "Codex CLI · effort medium",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            0,
            72.75
          ],
          "bRange": [
            0,
            71.43
          ],
          "calculation": true
        },
        {
          "metric": "Time to first text: a 250-line answer, six models",
          "aValue": 1.96,
          "bValue": 3.52,
          "unit": "seconds",
          "aDisplay": "1.96 s",
          "bDisplay": "3.52 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Sonnet 5.5 0.88 s to 4.09 s; GPT-6.1 Sol (Codex CLI) 2.75 s to 4.42 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "llm-speed-anatomy",
          "n": 4,
          "chartId": "speed-anatomy-first-text",
          "aN": 4,
          "bN": 4,
          "aContext": "Claude Code",
          "bContext": "Codex CLI · effort low",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            0.88,
            4.09
          ],
          "bRange": [
            2.75,
            4.42
          ]
        },
        {
          "metric": "Output speed after the first text: visible tokens per second (calculation)",
          "aValue": 231.7,
          "bValue": 79.6,
          "unit": "tokens",
          "aDisplay": "232",
          "bDisplay": "80",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "llm-speed-anatomy",
          "n": 4,
          "chartId": "speed-anatomy-output-speed",
          "aN": 4,
          "bN": 4,
          "aContext": "Claude Code",
          "bContext": "Codex CLI · effort low",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            230.3,
            233
          ],
          "bRange": [
            71.6,
            80.5
          ],
          "calculation": true
        },
        {
          "metric": "Output speed in characters per second after the first text (calculation)",
          "aValue": 517,
          "bValue": 323,
          "unit": "count",
          "aDisplay": "517",
          "bDisplay": "323",
          "winner": "unclear",
          "basis": "More or fewer count is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "llm-speed-anatomy",
          "n": 4,
          "chartId": "speed-anatomy-chars-per-second",
          "aN": 4,
          "bN": 4,
          "aContext": "Claude Code",
          "bContext": "Codex CLI · effort low",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            513,
            519
          ],
          "bRange": [
            291,
            327
          ],
          "calculation": true
        },
        {
          "metric": "Time to first text as the prompt grows: 1k",
          "aValue": 1.45,
          "bValue": 3.36,
          "unit": "seconds",
          "aDisplay": "1.45 s",
          "bDisplay": "3.36 s",
          "winner": "unclear",
          "basis": "Only 3 runs per side; the run ranges (fastest to slowest) do not overlap (Claude Sonnet 5.5 1.23 s to 1.72 s; GPT-6.1 Sol (Codex CLI) 3.36 s to 4.75 s), but 3 runs cannot show a reliable difference. A range is not a confidence interval.",
          "studySlug": "llm-speed-anatomy",
          "n": 3,
          "chartId": "speed-anatomy-prompt-size",
          "aN": 3,
          "bN": 3,
          "aContext": "Claude Code",
          "bContext": "Codex CLI · effort low",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            1.23,
            1.72
          ],
          "bRange": [
            3.36,
            4.75
          ],
          "calculation": true
        },
        {
          "metric": "Time to first text as the prompt grows: 16k",
          "aValue": 1.78,
          "bValue": 4.02,
          "unit": "seconds",
          "aDisplay": "1.78 s",
          "bDisplay": "4.02 s",
          "winner": "unclear",
          "basis": "Only 3 runs per side; the run ranges (fastest to slowest) do not overlap (Claude Sonnet 5.5 1.64 s to 2.11 s; GPT-6.1 Sol (Codex CLI) 3.30 s to 4.28 s), but 3 runs cannot show a reliable difference. A range is not a confidence interval.",
          "studySlug": "llm-speed-anatomy",
          "n": 3,
          "chartId": "speed-anatomy-prompt-size",
          "aN": 3,
          "bN": 3,
          "aContext": "Claude Code",
          "bContext": "Codex CLI · effort low",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            1.64,
            2.11
          ],
          "bRange": [
            3.3,
            4.28
          ],
          "calculation": true
        },
        {
          "metric": "Time to first text as the prompt grows: 64k",
          "aValue": 3.07,
          "bValue": 3.93,
          "unit": "seconds",
          "aDisplay": "3.07 s",
          "bDisplay": "3.93 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Sonnet 5.5 1.38 s to 3.61 s; GPT-6.1 Sol (Codex CLI) 3.42 s to 4.38 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "llm-speed-anatomy",
          "n": 3,
          "chartId": "speed-anatomy-prompt-size",
          "aN": 3,
          "bN": 3,
          "aContext": "Claude Code",
          "bContext": "Codex CLI · effort low",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            1.38,
            3.61
          ],
          "bRange": [
            3.42,
            4.38
          ],
          "calculation": true
        },
        {
          "metric": "Total time per call by prompt size (1k prompt)",
          "aValue": 1.78,
          "bValue": 3.43,
          "unit": "seconds",
          "aDisplay": "1.78 s",
          "bDisplay": "3.43 s",
          "winner": "unclear",
          "basis": "Only 3 runs per side; the run ranges (fastest to slowest) do not overlap (Claude Sonnet 5.5 1.57 s to 2.12 s; GPT-6.1 Sol (Codex CLI) 3.43 s to 4.92 s), but 3 runs cannot show a reliable difference. A range is not a confidence interval.",
          "studySlug": "llm-speed-anatomy",
          "n": 3,
          "chartId": "speed-anatomy-total-by-size",
          "aN": 3,
          "bN": 3,
          "aContext": "Claude Code",
          "bContext": "Codex CLI · effort low",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            1.57,
            2.12
          ],
          "bRange": [
            3.43,
            4.92
          ]
        },
        {
          "metric": "Total time per call by prompt size (16k prompt)",
          "aValue": 2.1,
          "bValue": 4.14,
          "unit": "seconds",
          "aDisplay": "2.10 s",
          "bDisplay": "4.14 s",
          "winner": "unclear",
          "basis": "Only 3 runs per side; the run ranges (fastest to slowest) do not overlap (Claude Sonnet 5.5 1.98 s to 2.48 s; GPT-6.1 Sol (Codex CLI) 3.96 s to 4.68 s), but 3 runs cannot show a reliable difference. A range is not a confidence interval.",
          "studySlug": "llm-speed-anatomy",
          "n": 3,
          "chartId": "speed-anatomy-total-by-size",
          "aN": 3,
          "bN": 3,
          "aContext": "Claude Code",
          "bContext": "Codex CLI · effort low",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            1.98,
            2.48
          ],
          "bRange": [
            3.96,
            4.68
          ]
        },
        {
          "metric": "Total time per call by prompt size (64k prompt)",
          "aValue": 3.44,
          "bValue": 3.96,
          "unit": "seconds",
          "aDisplay": "3.44 s",
          "bDisplay": "3.96 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Sonnet 5.5 1.74 s to 4.38 s; GPT-6.1 Sol (Codex CLI) 3.47 s to 4.44 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "llm-speed-anatomy",
          "n": 3,
          "chartId": "speed-anatomy-total-by-size",
          "aN": 3,
          "bN": 3,
          "aContext": "Claude Code",
          "bContext": "Codex CLI · effort low",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            1.74,
            4.38
          ],
          "bRange": [
            3.47,
            4.44
          ]
        },
        {
          "metric": "Exact lookup answers at the 1k, 16k and 64k prompt-size targets",
          "aValue": 1,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "100% (9/9)",
          "bDisplay": "100% (9/9)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Sonnet 5.5 70% to 100%; GPT-6.1 Sol (Codex CLI) 70% to 100%), so this sample cannot separate them.",
          "studySlug": "llm-speed-anatomy",
          "n": 9,
          "chartId": "speed-anatomy-lookup-correct",
          "aN": 9,
          "bN": 9,
          "aContext": "Claude Code",
          "bContext": "Codex CLI · effort low",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.7009,
            1
          ],
          "bRange": [
            0.7009,
            1
          ]
        },
        {
          "metric": "Pass rate on 4 harder tasks (Strict pass)",
          "aValue": 0.375,
          "bValue": 0.6875,
          "unit": "rate",
          "aDisplay": "38% (6/16)",
          "bDisplay": "69% (11/16)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Sonnet 5.5 18% to 61%; GPT-6.1 Sol (Codex CLI) 44% to 86%), so this sample cannot separate them.",
          "studySlug": "harder-tasks-head-to-head",
          "n": 16,
          "chartId": "harder-h2h-pass-rate",
          "aN": 16,
          "bN": 16,
          "aContext": "Claude Code",
          "bContext": "Codex CLI · effort medium",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.1848,
            0.6136
          ],
          "bRange": [
            0.444,
            0.8584
          ]
        },
        {
          "metric": "Pass rate on 4 harder tasks (Lenient (format misses counted))",
          "aValue": 0.375,
          "bValue": 0.6875,
          "unit": "rate",
          "aDisplay": "38% (6/16)",
          "bDisplay": "69% (11/16)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Sonnet 5.5 18% to 61%; GPT-6.1 Sol (Codex CLI) 44% to 86%), so this sample cannot separate them.",
          "studySlug": "harder-tasks-head-to-head",
          "n": 16,
          "chartId": "harder-h2h-pass-rate",
          "aN": 16,
          "bN": 16,
          "aContext": "Claude Code",
          "bContext": "Codex CLI · effort medium",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.1848,
            0.6136
          ],
          "bRange": [
            0.444,
            0.8584
          ]
        },
        {
          "metric": "Calls that tried a tool although tools were off",
          "aValue": 0.3125,
          "bValue": 0,
          "unit": "rate",
          "aDisplay": "31% (5/16)",
          "bDisplay": "0% (0/16)",
          "winner": "unclear",
          "basis": "More or fewer rate is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "harder-tasks-head-to-head",
          "n": 16,
          "chartId": "harder-h2h-tool-attempts",
          "aN": 16,
          "bN": 16,
          "aContext": "Claude Code",
          "bContext": "Codex CLI · effort medium",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.1416,
            0.556
          ],
          "bRange": [
            0,
            0.1936
          ]
        },
        {
          "metric": "Strict pass rate by task: 10x10 nonogram",
          "aValue": 1,
          "bValue": 0.75,
          "unit": "rate",
          "aDisplay": "100% (4/4)",
          "bDisplay": "75% (3/4)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Sonnet 5.5 51% to 100%; GPT-6.1 Sol (Codex CLI) 30% to 95%), so this sample cannot separate them.",
          "studySlug": "harder-tasks-head-to-head",
          "n": 4,
          "chartId": "harder-h2h-pass-by-task",
          "aN": 4,
          "bN": 4,
          "aContext": "Claude Code",
          "bContext": "Codex CLI · effort medium",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.5101,
            1
          ],
          "bRange": [
            0.3006,
            0.9544
          ]
        },
        {
          "metric": "Strict pass rate by task: Sudoku, 22 givens",
          "aValue": 0,
          "bValue": 0.25,
          "unit": "rate",
          "aDisplay": "0% (0/4)",
          "bDisplay": "25% (1/4)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Sonnet 5.5 0% to 49%; GPT-6.1 Sol (Codex CLI) 5% to 70%), so this sample cannot separate them.",
          "studySlug": "harder-tasks-head-to-head",
          "n": 4,
          "chartId": "harder-h2h-pass-by-task",
          "aN": 4,
          "bN": 4,
          "aContext": "Claude Code",
          "bContext": "Codex CLI · effort medium",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0,
            0.4899
          ],
          "bRange": [
            0.0456,
            0.6994
          ]
        },
        {
          "metric": "Strict pass rate by task: 6x6 Skyscrapers",
          "aValue": 0,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "0% (0/4)",
          "bDisplay": "100% (4/4)",
          "winner": "b",
          "basis": "The 95% intervals do not overlap (Claude Sonnet 5.5 0% to 49%; GPT-6.1 Sol (Codex CLI) 51% to 100%).",
          "studySlug": "harder-tasks-head-to-head",
          "n": 4,
          "chartId": "harder-h2h-pass-by-task",
          "aN": 4,
          "bN": 4,
          "aContext": "Claude Code",
          "bContext": "Codex CLI · effort medium",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0,
            0.4899
          ],
          "bRange": [
            0.5101,
            1
          ]
        },
        {
          "metric": "Strict pass rate by task: Seeded shuffle output",
          "aValue": 0.5,
          "bValue": 0.75,
          "unit": "rate",
          "aDisplay": "50% (2/4)",
          "bDisplay": "75% (3/4)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Sonnet 5.5 15% to 85%; GPT-6.1 Sol (Codex CLI) 30% to 95%), so this sample cannot separate them.",
          "studySlug": "harder-tasks-head-to-head",
          "n": 4,
          "chartId": "harder-h2h-pass-by-task",
          "aN": 4,
          "bN": 4,
          "aContext": "Claude Code",
          "bContext": "Codex CLI · effort medium",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.15,
            0.85
          ],
          "bRange": [
            0.3006,
            0.9544
          ]
        },
        {
          "metric": "Total time per call on harder tasks",
          "aValue": 70.43,
          "bValue": 120.24,
          "unit": "seconds",
          "aDisplay": "70.4 s",
          "bDisplay": "120.2 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Sonnet 5.5 4.32 s to 210.1 s; GPT-6.1 Sol (Codex CLI) 46.2 s to 273.5 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-total-latency",
          "aN": 12,
          "bN": 13,
          "aContext": "Claude Code",
          "bContext": "Codex CLI · effort medium",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            4.32,
            210.08
          ],
          "bRange": [
            46.24,
            273.46
          ]
        },
        {
          "metric": "Output tokens per call on harder tasks (Output tokens)",
          "aValue": 9287,
          "bValue": 4994,
          "unit": "tokens",
          "aDisplay": "9,287",
          "bDisplay": "4,994",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-output-tokens",
          "aN": 12,
          "bN": 13,
          "aContext": "Claude Code",
          "bContext": "Codex CLI · effort medium",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            407,
            27921
          ],
          "bRange": [
            2099,
            13413
          ]
        },
        {
          "metric": "List-price cost per strict pass on harder tasks (calculation)",
          "aValue": 0.23843,
          "bValue": 0.08293,
          "unit": "usd",
          "aDisplay": "$0.24",
          "bDisplay": "$0.083",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.24 vs $0.083, 2.9x) is not tested against run-to-run variation.",
          "studySlug": "harder-tasks-head-to-head",
          "n": 16,
          "chartId": "harder-h2h-cost-per-pass",
          "aN": 16,
          "bN": 16,
          "aContext": "Claude Code",
          "bContext": "Codex CLI · effort medium",
          "calculation": true
        }
      ]
    },
    {
      "slug": "jev-1-13-vs-claude-haiku-4-5",
      "a": "jev-1-13",
      "b": "claude-haiku-4-5",
      "title": "Jev 1.13 vs Claude Haiku 4.5",
      "seoTitle": "Jev 1.13 vs Claude Haiku 4.5: measured benchmarks",
      "description": "Jev 1.13 vs Claude Haiku 4.5: 17 measured metrics from 3 studies (Typed routing decisions answered exactly right; more), with sample sizes and intervals.",
      "verdict": "Jev 1.13 and Claude Haiku 4.5 share 17 measured metrics and 6 list-price calculations from 3 studies. Jev 1.13 leads on 2 rows: Time to make one routing decision, 137 ms vs 12,543 ms; Time per routing decision, by route (Wall time), 0.14 s vs 9.44 s. On those rows the p50–p95 bands do not overlap; only a 95% interval is a confidence interval. The other rows are 15 ties and 6 unclear; each row says why. Calculation rows are derived from list prices and recorded counts; they are not bills or runs. 13 rows ran the two sides through different routes (for example TypeSafe API vs Claude Code), so they compare route + model pairs, not models alone; the contexts name the route. Some rows rest on small samples (n = 12 at the smallest).",
      "rows": [
        {
          "metric": "Typed routing decisions answered exactly right",
          "aValue": 0.8984,
          "bValue": 0.8902,
          "unit": "rate",
          "aDisplay": "90%",
          "bDisplay": "89% (73/82)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Jev 1.13 82% to 95%; Claude Haiku 4.5 80% to 94%), so this sample cannot separate them.",
          "studySlug": "routing-jev-vs-llm",
          "n": 82,
          "chartId": "routing-exact-decisions",
          "aN": 82,
          "bN": 82,
          "aContext": "typed routing decisions · TypeSafe API",
          "bContext": "typed routing decisions · via Claude Code",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.8191,
            0.9497
          ],
          "bRange": [
            0.8044,
            0.9412
          ]
        },
        {
          "metric": "Per-question accuracy",
          "aValue": 0.9485,
          "bValue": 0.9433,
          "unit": "rate",
          "aDisplay": "95% (184/194)",
          "bDisplay": "94% (183/194)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Jev 1.13 91% to 97%; Claude Haiku 4.5 90% to 97%), so this sample cannot separate them.",
          "studySlug": "routing-jev-vs-llm",
          "n": 194,
          "chartId": "routing-key-accuracy",
          "aN": 194,
          "bN": 194,
          "aContext": "typed routing decisions · TypeSafe API",
          "bContext": "typed routing decisions · via Claude Code",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.9077,
            0.9718
          ],
          "bRange": [
            0.9013,
            0.968
          ]
        },
        {
          "metric": "Exact rate by decision type: Failure class",
          "aValue": 1,
          "bValue": 0.9444,
          "unit": "rate",
          "aDisplay": "100% (18/18)",
          "bDisplay": "94% (17/18)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Jev 1.13 82% to 100%; Claude Haiku 4.5 74% to 99%), so this sample cannot separate them.",
          "studySlug": "routing-jev-vs-llm",
          "n": 18,
          "chartId": "routing-exact-by-decision",
          "aN": 18,
          "bN": 18,
          "aContext": "typed routing decisions · TypeSafe API",
          "bContext": "typed routing decisions · via Claude Code",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.8241,
            1
          ],
          "bRange": [
            0.7424,
            0.9901
          ]
        },
        {
          "metric": "Exact rate by decision type: Message intent",
          "aValue": 1,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "100% (20/20)",
          "bDisplay": "100% (20/20)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Jev 1.13 84% to 100%; Claude Haiku 4.5 84% to 100%), so this sample cannot separate them.",
          "studySlug": "routing-jev-vs-llm",
          "n": 20,
          "chartId": "routing-exact-by-decision",
          "aN": 20,
          "bN": 20,
          "aContext": "typed routing decisions · TypeSafe API",
          "bContext": "typed routing decisions · via Claude Code",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.8389,
            1
          ],
          "bRange": [
            0.8389,
            1
          ]
        },
        {
          "metric": "Exact rate by decision type: Is it a rule?",
          "aValue": 1,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "100% (12/12)",
          "bDisplay": "100% (12/12)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Jev 1.13 76% to 100%; Claude Haiku 4.5 76% to 100%), so this sample cannot separate them.",
          "studySlug": "routing-jev-vs-llm",
          "n": 12,
          "chartId": "routing-exact-by-decision",
          "aN": 12,
          "bN": 12,
          "aContext": "typed routing decisions · TypeSafe API",
          "bContext": "typed routing decisions · via Claude Code",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.7575,
            1
          ],
          "bRange": [
            0.7575,
            1
          ]
        },
        {
          "metric": "Exact rate by decision type: Context shape",
          "aValue": 0.7396,
          "bValue": 0.75,
          "unit": "rate",
          "aDisplay": "74%",
          "bDisplay": "75% (24/32)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Jev 1.13 58% to 87%; Claude Haiku 4.5 58% to 87%), so this sample cannot separate them.",
          "studySlug": "routing-jev-vs-llm",
          "n": 32,
          "chartId": "routing-exact-by-decision",
          "aN": 32,
          "bN": 32,
          "aContext": "typed routing decisions · TypeSafe API",
          "bContext": "typed routing decisions · via Claude Code",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.5789,
            0.8675
          ],
          "bRange": [
            0.5789,
            0.8675
          ]
        },
        {
          "metric": "Cost per 1,000 routing decisions",
          "aValue": 0.0337,
          "bValue": 8.924,
          "unit": "usd",
          "aDisplay": "$0.034",
          "bDisplay": "$8.92",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.034 vs $8.92, 265x) is not tested against run-to-run variation.",
          "studySlug": "routing-jev-vs-llm",
          "chartId": "routing-cost-per-1000",
          "aN": 246,
          "bN": 82,
          "aContext": "typed routing decisions · TypeSafe API",
          "bContext": "typed routing decisions · via Claude Code",
          "calculation": true
        },
        {
          "metric": "Time to make one routing decision",
          "aValue": 136.5,
          "bValue": 12543,
          "unit": "ms",
          "aDisplay": "137 ms",
          "bDisplay": "12,543 ms",
          "winner": "a",
          "basis": "Claude Haiku 4.5’s median is above Jev 1.13’s 95th percentile (p50–p95 bands: Jev 1.13 137 ms to 196 ms; Claude Haiku 4.5 12,543 ms to 34,481 ms); not a confidence interval. The two sides ran through different routes (TypeSafe API vs Claude Code), so this row compares routes, not models alone.",
          "studySlug": "routing-overhead",
          "chartId": "router-overhead-decision-latency",
          "aN": 246,
          "bN": 82,
          "aContext": "routing overhead per decision · TypeSafe API",
          "bContext": "thinking on · via Claude Code · routing overhead per decision",
          "rangeKind": "range",
          "spanKind": "p50-p95",
          "aRange": [
            136.5,
            195.7
          ],
          "bRange": [
            12543,
            34481
          ]
        },
        {
          "metric": "Routing calls that returned a decision",
          "aValue": 1,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "100% (246/246)",
          "bDisplay": "100% (82/82)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Jev 1.13 98% to 100%; Claude Haiku 4.5 96% to 100%), so this sample cannot separate them.",
          "studySlug": "routing-overhead",
          "chartId": "router-overhead-completed",
          "aN": 246,
          "bN": 82,
          "aContext": "routing overhead per decision · TypeSafe API",
          "bContext": "thinking on · via Claude Code · routing overhead per decision",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.9846,
            1
          ],
          "bRange": [
            0.9552,
            1
          ]
        },
        {
          "metric": "Added routing cost per 1,000 tasks (calculation) (Every model call routed (49.5 per task))",
          "aValue": 1.67,
          "bValue": 441.74,
          "unit": "usd",
          "aDisplay": "$1.67",
          "bDisplay": "$441.74",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($1.67 vs $441.74, 265x) is not tested against run-to-run variation.",
          "studySlug": "routing-overhead",
          "chartId": "router-overhead-cost-per-1000-tasks",
          "aContext": "calculation per 1,000 tasks from recorded decision counts · TypeSafe API",
          "bContext": "thinking on · via Claude Code · calculation per 1,000 tasks from recorded decision counts",
          "calculation": true
        },
        {
          "metric": "Added routing cost per 1,000 tasks (calculation) (Only System One decisions (7 per task))",
          "aValue": 0.24,
          "bValue": 62.47,
          "unit": "usd",
          "aDisplay": "$0.24",
          "bDisplay": "$62.47",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.24 vs $62.47, 260x) is not tested against run-to-run variation.",
          "studySlug": "routing-overhead",
          "chartId": "router-overhead-cost-per-1000-tasks",
          "aContext": "calculation per 1,000 tasks from recorded decision counts · TypeSafe API",
          "bContext": "thinking on · via Claude Code · calculation per 1,000 tasks from recorded decision counts",
          "calculation": true
        },
        {
          "metric": "Added routing delay per task (calculation) (Every model call routed (49.5 per task))",
          "aValue": 6.7568,
          "bValue": 620.8785,
          "unit": "seconds",
          "aDisplay": "6.76 s",
          "bDisplay": "620.9 s",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap (6.76 s vs 620.9 s, 92x) is not tested against run-to-run variation.",
          "studySlug": "routing-overhead",
          "chartId": "router-overhead-delay-per-task",
          "aContext": "calculation per task from recorded decision counts, decisions in line · TypeSafe API",
          "bContext": "thinking on · via Claude Code · calculation per task from recorded decision counts, decisions in line",
          "calculation": true
        },
        {
          "metric": "Added routing delay per task (calculation) (Only System One decisions (7 per task))",
          "aValue": 0.9555,
          "bValue": 87.801,
          "unit": "seconds",
          "aDisplay": "0.96 s",
          "bDisplay": "87.8 s",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap (0.96 s vs 87.8 s, 92x) is not tested against run-to-run variation.",
          "studySlug": "routing-overhead",
          "chartId": "router-overhead-delay-per-task",
          "aContext": "calculation per task from recorded decision counts, decisions in line · TypeSafe API",
          "bContext": "thinking on · via Claude Code · calculation per task from recorded decision counts, decisions in line",
          "calculation": true
        },
        {
          "metric": "Unseen routing decisions answered exactly right",
          "aValue": 0.8214,
          "bValue": 0.7857,
          "unit": "rate",
          "aDisplay": "82% (46/56)",
          "bDisplay": "79% (44/56)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Jev 1.13 70% to 90%; Claude Haiku 4.5 66% to 87%), so this sample cannot separate them.",
          "studySlug": "routing-holdout",
          "n": 56,
          "chartId": "routing-holdout-exact",
          "aN": 56,
          "bN": 56,
          "aContext": "",
          "bContext": "Claude Code",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.7016,
            0.9
          ],
          "bRange": [
            0.6618,
            0.8729
          ]
        },
        {
          "metric": "Per-question accuracy on unseen decisions",
          "aValue": 0.904,
          "bValue": 0.816,
          "unit": "rate",
          "aDisplay": "90% (113/125)",
          "bDisplay": "82% (102/125)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Jev 1.13 84% to 94%; Claude Haiku 4.5 74% to 87%), so this sample cannot separate them.",
          "studySlug": "routing-holdout",
          "n": 125,
          "chartId": "routing-holdout-key-accuracy",
          "aN": 125,
          "bN": 125,
          "aContext": "",
          "bContext": "Claude Code",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.8397,
            0.9442
          ],
          "bRange": [
            0.739,
            0.8741
          ]
        },
        {
          "metric": "Exact rate on unseen decisions, by decision type: Failure class",
          "aValue": 0.9286,
          "bValue": 0.9286,
          "unit": "rate",
          "aDisplay": "93% (13/14)",
          "bDisplay": "93% (13/14)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Jev 1.13 69% to 99%; Claude Haiku 4.5 69% to 99%), so this sample cannot separate them.",
          "studySlug": "routing-holdout",
          "n": 14,
          "chartId": "routing-holdout-by-purpose",
          "aN": 14,
          "bN": 14,
          "aContext": "",
          "bContext": "Claude Code",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.6853,
            0.9873
          ],
          "bRange": [
            0.6853,
            0.9873
          ]
        },
        {
          "metric": "Exact rate on unseen decisions, by decision type: Message intent",
          "aValue": 0.8571,
          "bValue": 0.9286,
          "unit": "rate",
          "aDisplay": "86% (12/14)",
          "bDisplay": "93% (13/14)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Jev 1.13 60% to 96%; Claude Haiku 4.5 69% to 99%), so this sample cannot separate them.",
          "studySlug": "routing-holdout",
          "n": 14,
          "chartId": "routing-holdout-by-purpose",
          "aN": 14,
          "bN": 14,
          "aContext": "",
          "bContext": "Claude Code",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.6006,
            0.9599
          ],
          "bRange": [
            0.6853,
            0.9873
          ]
        },
        {
          "metric": "Exact rate on unseen decisions, by decision type: Is it a rule?",
          "aValue": 0.9286,
          "bValue": 0.9286,
          "unit": "rate",
          "aDisplay": "93% (13/14)",
          "bDisplay": "93% (13/14)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Jev 1.13 69% to 99%; Claude Haiku 4.5 69% to 99%), so this sample cannot separate them.",
          "studySlug": "routing-holdout",
          "n": 14,
          "chartId": "routing-holdout-by-purpose",
          "aN": 14,
          "bN": 14,
          "aContext": "",
          "bContext": "Claude Code",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.6853,
            0.9873
          ],
          "bRange": [
            0.6853,
            0.9873
          ]
        },
        {
          "metric": "Exact rate on unseen decisions, by decision type: Context shape",
          "aValue": 0.5714,
          "bValue": 0.3571,
          "unit": "rate",
          "aDisplay": "57% (8/14)",
          "bDisplay": "36% (5/14)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Jev 1.13 33% to 79%; Claude Haiku 4.5 16% to 61%), so this sample cannot separate them.",
          "studySlug": "routing-holdout",
          "n": 14,
          "chartId": "routing-holdout-by-purpose",
          "aN": 14,
          "bN": 14,
          "aContext": "",
          "bContext": "Claude Code",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.3259,
            0.7862
          ],
          "bRange": [
            0.1634,
            0.6124
          ]
        },
        {
          "metric": "Tuned case set vs unseen holdout: exact rate per router (Tuned set (routing-jev-vs-llm))",
          "aValue": 0.9024,
          "bValue": 0.8902,
          "unit": "rate",
          "aDisplay": "90% (74/82)",
          "bDisplay": "89% (73/82)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Jev 1.13 82% to 95%; Claude Haiku 4.5 80% to 94%), so this sample cannot separate them.",
          "studySlug": "routing-holdout",
          "n": 82,
          "chartId": "routing-holdout-tuned-vs-unseen",
          "aN": 82,
          "bN": 82,
          "aContext": "",
          "bContext": "Claude Code",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.8191,
            0.9497
          ],
          "bRange": [
            0.8044,
            0.9412
          ]
        },
        {
          "metric": "Tuned case set vs unseen holdout: exact rate per router (Unseen holdout)",
          "aValue": 0.8214,
          "bValue": 0.7857,
          "unit": "rate",
          "aDisplay": "82% (46/56)",
          "bDisplay": "79% (44/56)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Jev 1.13 70% to 90%; Claude Haiku 4.5 66% to 87%), so this sample cannot separate them.",
          "studySlug": "routing-holdout",
          "n": 56,
          "chartId": "routing-holdout-tuned-vs-unseen",
          "aN": 56,
          "bN": 56,
          "aContext": "",
          "bContext": "Claude Code",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.7016,
            0.9
          ],
          "bRange": [
            0.6618,
            0.8729
          ]
        },
        {
          "metric": "Time per routing decision, by route (Wall time)",
          "aValue": 0.139,
          "bValue": 9.444,
          "unit": "seconds",
          "aDisplay": "0.14 s",
          "bDisplay": "9.44 s",
          "winner": "a",
          "basis": "Claude Haiku 4.5’s median is above Jev 1.13’s 95th percentile (p50–p95 bands: Jev 1.13 0.14 s to 0.19 s; Claude Haiku 4.5 9.44 s to 25.4 s); not a confidence interval.",
          "studySlug": "routing-holdout",
          "chartId": "routing-holdout-latency",
          "aN": 168,
          "bN": 56,
          "aContext": "",
          "bContext": "Claude Code",
          "rangeKind": "range",
          "spanKind": "p50-p95",
          "aRange": [
            0.139,
            0.192
          ],
          "bRange": [
            9.444,
            25.413
          ]
        },
        {
          "metric": "Cost per 1,000 unseen routing decisions",
          "aValue": 0.03065,
          "bValue": 7.129,
          "unit": "usd",
          "aDisplay": "$0.031",
          "bDisplay": "$7.13",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.031 vs $7.13, 233x) is not tested against run-to-run variation.",
          "studySlug": "routing-holdout",
          "chartId": "routing-holdout-cost-per-1000",
          "aN": 168,
          "bN": 56,
          "aContext": "",
          "bContext": "Claude Code",
          "calculation": true
        }
      ]
    },
    {
      "slug": "jev-1-13-vs-claude-sonnet-5-5",
      "a": "jev-1-13",
      "b": "claude-sonnet-5-5",
      "title": "Jev 1.13 vs Claude Sonnet 5.5",
      "seoTitle": "Jev 1.13 vs Claude Sonnet 5.5: measured benchmarks",
      "description": "Jev 1.13 vs Claude Sonnet 5.5: 17 measured metrics from 3 studies (Typed routing decisions answered exactly right; more), with sample sizes and intervals.",
      "verdict": "Jev 1.13 and Claude Sonnet 5.5 share 17 measured metrics and 6 list-price calculations from 3 studies. Jev 1.13 leads on 2 rows: Time to make one routing decision, 137 ms vs 2,597 ms; Time per routing decision, by route (Wall time), 0.14 s vs 2.36 s. On those rows the p50–p95 bands do not overlap; only a 95% interval is a confidence interval. The other rows are 15 ties and 6 unclear; each row says why. Calculation rows are derived from list prices and recorded counts; they are not bills or runs. 13 rows ran the two sides through different routes (for example TypeSafe API vs Claude Code), so they compare route + model pairs, not models alone; the contexts name the route. Some rows rest on small samples (n = 12 at the smallest).",
      "rows": [
        {
          "metric": "Typed routing decisions answered exactly right",
          "aValue": 0.8984,
          "bValue": 0.939,
          "unit": "rate",
          "aDisplay": "90%",
          "bDisplay": "94% (77/82)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Jev 1.13 82% to 95%; Claude Sonnet 5.5 87% to 97%), so this sample cannot separate them.",
          "studySlug": "routing-jev-vs-llm",
          "n": 82,
          "chartId": "routing-exact-decisions",
          "aN": 82,
          "bN": 82,
          "aContext": "typed routing decisions · TypeSafe API",
          "bContext": "typed routing decisions · via Claude Code",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.8191,
            0.9497
          ],
          "bRange": [
            0.8651,
            0.9737
          ]
        },
        {
          "metric": "Per-question accuracy",
          "aValue": 0.9485,
          "bValue": 0.9742,
          "unit": "rate",
          "aDisplay": "95% (184/194)",
          "bDisplay": "97% (189/194)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Jev 1.13 91% to 97%; Claude Sonnet 5.5 94% to 99%), so this sample cannot separate them.",
          "studySlug": "routing-jev-vs-llm",
          "n": 194,
          "chartId": "routing-key-accuracy",
          "aN": 194,
          "bN": 194,
          "aContext": "typed routing decisions · TypeSafe API",
          "bContext": "typed routing decisions · via Claude Code",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.9077,
            0.9718
          ],
          "bRange": [
            0.9411,
            0.9889
          ]
        },
        {
          "metric": "Exact rate by decision type: Failure class",
          "aValue": 1,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "100% (18/18)",
          "bDisplay": "100% (18/18)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Jev 1.13 82% to 100%; Claude Sonnet 5.5 82% to 100%), so this sample cannot separate them.",
          "studySlug": "routing-jev-vs-llm",
          "n": 18,
          "chartId": "routing-exact-by-decision",
          "aN": 18,
          "bN": 18,
          "aContext": "typed routing decisions · TypeSafe API",
          "bContext": "typed routing decisions · via Claude Code",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.8241,
            1
          ],
          "bRange": [
            0.8241,
            1
          ]
        },
        {
          "metric": "Exact rate by decision type: Message intent",
          "aValue": 1,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "100% (20/20)",
          "bDisplay": "100% (20/20)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Jev 1.13 84% to 100%; Claude Sonnet 5.5 84% to 100%), so this sample cannot separate them.",
          "studySlug": "routing-jev-vs-llm",
          "n": 20,
          "chartId": "routing-exact-by-decision",
          "aN": 20,
          "bN": 20,
          "aContext": "typed routing decisions · TypeSafe API",
          "bContext": "typed routing decisions · via Claude Code",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.8389,
            1
          ],
          "bRange": [
            0.8389,
            1
          ]
        },
        {
          "metric": "Exact rate by decision type: Is it a rule?",
          "aValue": 1,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "100% (12/12)",
          "bDisplay": "100% (12/12)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Jev 1.13 76% to 100%; Claude Sonnet 5.5 76% to 100%), so this sample cannot separate them.",
          "studySlug": "routing-jev-vs-llm",
          "n": 12,
          "chartId": "routing-exact-by-decision",
          "aN": 12,
          "bN": 12,
          "aContext": "typed routing decisions · TypeSafe API",
          "bContext": "typed routing decisions · via Claude Code",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.7575,
            1
          ],
          "bRange": [
            0.7575,
            1
          ]
        },
        {
          "metric": "Exact rate by decision type: Context shape",
          "aValue": 0.7396,
          "bValue": 0.8438,
          "unit": "rate",
          "aDisplay": "74%",
          "bDisplay": "84% (27/32)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Jev 1.13 58% to 87%; Claude Sonnet 5.5 68% to 93%), so this sample cannot separate them.",
          "studySlug": "routing-jev-vs-llm",
          "n": 32,
          "chartId": "routing-exact-by-decision",
          "aN": 32,
          "bN": 32,
          "aContext": "typed routing decisions · TypeSafe API",
          "bContext": "typed routing decisions · via Claude Code",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.5789,
            0.8675
          ],
          "bRange": [
            0.6825,
            0.9314
          ]
        },
        {
          "metric": "Cost per 1,000 routing decisions",
          "aValue": 0.0337,
          "bValue": 4.996,
          "unit": "usd",
          "aDisplay": "$0.034",
          "bDisplay": "$5.00",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.034 vs $5.00, 148x) is not tested against run-to-run variation.",
          "studySlug": "routing-jev-vs-llm",
          "chartId": "routing-cost-per-1000",
          "aN": 246,
          "bN": 82,
          "aContext": "typed routing decisions · TypeSafe API",
          "bContext": "typed routing decisions · via Claude Code",
          "calculation": true
        },
        {
          "metric": "Time to make one routing decision",
          "aValue": 136.5,
          "bValue": 2597,
          "unit": "ms",
          "aDisplay": "137 ms",
          "bDisplay": "2,597 ms",
          "winner": "a",
          "basis": "Claude Sonnet 5.5’s median is above Jev 1.13’s 95th percentile (p50–p95 bands: Jev 1.13 137 ms to 196 ms; Claude Sonnet 5.5 2,597 ms to 4,298 ms); not a confidence interval. The two sides ran through different routes (TypeSafe API vs Claude Code), so this row compares routes, not models alone.",
          "studySlug": "routing-overhead",
          "chartId": "router-overhead-decision-latency",
          "aN": 246,
          "bN": 82,
          "aContext": "routing overhead per decision · TypeSafe API",
          "bContext": "effort low · via Claude Code · routing overhead per decision",
          "rangeKind": "range",
          "spanKind": "p50-p95",
          "aRange": [
            136.5,
            195.7
          ],
          "bRange": [
            2597,
            4298
          ]
        },
        {
          "metric": "Routing calls that returned a decision",
          "aValue": 1,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "100% (246/246)",
          "bDisplay": "100% (82/82)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Jev 1.13 98% to 100%; Claude Sonnet 5.5 96% to 100%), so this sample cannot separate them.",
          "studySlug": "routing-overhead",
          "chartId": "router-overhead-completed",
          "aN": 246,
          "bN": 82,
          "aContext": "routing overhead per decision · TypeSafe API",
          "bContext": "effort low · via Claude Code · routing overhead per decision",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.9846,
            1
          ],
          "bRange": [
            0.9552,
            1
          ]
        },
        {
          "metric": "Added routing cost per 1,000 tasks (calculation) (Every model call routed (49.5 per task))",
          "aValue": 1.67,
          "bValue": 247.3,
          "unit": "usd",
          "aDisplay": "$1.67",
          "bDisplay": "$247.30",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($1.67 vs $247.30, 148x) is not tested against run-to-run variation.",
          "studySlug": "routing-overhead",
          "chartId": "router-overhead-cost-per-1000-tasks",
          "aContext": "calculation per 1,000 tasks from recorded decision counts · TypeSafe API",
          "bContext": "effort low · via Claude Code · calculation per 1,000 tasks from recorded decision counts",
          "calculation": true
        },
        {
          "metric": "Added routing cost per 1,000 tasks (calculation) (Only System One decisions (7 per task))",
          "aValue": 0.24,
          "bValue": 34.97,
          "unit": "usd",
          "aDisplay": "$0.24",
          "bDisplay": "$34.97",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.24 vs $34.97, 146x) is not tested against run-to-run variation.",
          "studySlug": "routing-overhead",
          "chartId": "router-overhead-cost-per-1000-tasks",
          "aContext": "calculation per 1,000 tasks from recorded decision counts · TypeSafe API",
          "bContext": "effort low · via Claude Code · calculation per 1,000 tasks from recorded decision counts",
          "calculation": true
        },
        {
          "metric": "Added routing delay per task (calculation) (Every model call routed (49.5 per task))",
          "aValue": 6.7568,
          "bValue": 128.5515,
          "unit": "seconds",
          "aDisplay": "6.76 s",
          "bDisplay": "128.6 s",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap (6.76 s vs 128.6 s, 19x) is not tested against run-to-run variation.",
          "studySlug": "routing-overhead",
          "chartId": "router-overhead-delay-per-task",
          "aContext": "calculation per task from recorded decision counts, decisions in line · TypeSafe API",
          "bContext": "effort low · via Claude Code · calculation per task from recorded decision counts, decisions in line",
          "calculation": true
        },
        {
          "metric": "Added routing delay per task (calculation) (Only System One decisions (7 per task))",
          "aValue": 0.9555,
          "bValue": 18.179,
          "unit": "seconds",
          "aDisplay": "0.96 s",
          "bDisplay": "18.2 s",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap (0.96 s vs 18.2 s, 19x) is not tested against run-to-run variation.",
          "studySlug": "routing-overhead",
          "chartId": "router-overhead-delay-per-task",
          "aContext": "calculation per task from recorded decision counts, decisions in line · TypeSafe API",
          "bContext": "effort low · via Claude Code · calculation per task from recorded decision counts, decisions in line",
          "calculation": true
        },
        {
          "metric": "Unseen routing decisions answered exactly right",
          "aValue": 0.8214,
          "bValue": 0.875,
          "unit": "rate",
          "aDisplay": "82% (46/56)",
          "bDisplay": "88% (49/56)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Jev 1.13 70% to 90%; Claude Sonnet 5.5 76% to 94%), so this sample cannot separate them.",
          "studySlug": "routing-holdout",
          "n": 56,
          "chartId": "routing-holdout-exact",
          "aN": 56,
          "bN": 56,
          "aContext": "",
          "bContext": "Claude Code · effort low",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.7016,
            0.9
          ],
          "bRange": [
            0.7637,
            0.9381
          ]
        },
        {
          "metric": "Per-question accuracy on unseen decisions",
          "aValue": 0.904,
          "bValue": 0.92,
          "unit": "rate",
          "aDisplay": "90% (113/125)",
          "bDisplay": "92% (115/125)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Jev 1.13 84% to 94%; Claude Sonnet 5.5 86% to 96%), so this sample cannot separate them.",
          "studySlug": "routing-holdout",
          "n": 125,
          "chartId": "routing-holdout-key-accuracy",
          "aN": 125,
          "bN": 125,
          "aContext": "",
          "bContext": "Claude Code · effort low",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.8397,
            0.9442
          ],
          "bRange": [
            0.859,
            0.956
          ]
        },
        {
          "metric": "Exact rate on unseen decisions, by decision type: Failure class",
          "aValue": 0.9286,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "93% (13/14)",
          "bDisplay": "100% (14/14)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Jev 1.13 69% to 99%; Claude Sonnet 5.5 78% to 100%), so this sample cannot separate them.",
          "studySlug": "routing-holdout",
          "n": 14,
          "chartId": "routing-holdout-by-purpose",
          "aN": 14,
          "bN": 14,
          "aContext": "",
          "bContext": "Claude Code · effort low",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.6853,
            0.9873
          ],
          "bRange": [
            0.7847,
            1
          ]
        },
        {
          "metric": "Exact rate on unseen decisions, by decision type: Message intent",
          "aValue": 0.8571,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "86% (12/14)",
          "bDisplay": "100% (14/14)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Jev 1.13 60% to 96%; Claude Sonnet 5.5 78% to 100%), so this sample cannot separate them.",
          "studySlug": "routing-holdout",
          "n": 14,
          "chartId": "routing-holdout-by-purpose",
          "aN": 14,
          "bN": 14,
          "aContext": "",
          "bContext": "Claude Code · effort low",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.6006,
            0.9599
          ],
          "bRange": [
            0.7847,
            1
          ]
        },
        {
          "metric": "Exact rate on unseen decisions, by decision type: Is it a rule?",
          "aValue": 0.9286,
          "bValue": 0.9286,
          "unit": "rate",
          "aDisplay": "93% (13/14)",
          "bDisplay": "93% (13/14)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Jev 1.13 69% to 99%; Claude Sonnet 5.5 69% to 99%), so this sample cannot separate them.",
          "studySlug": "routing-holdout",
          "n": 14,
          "chartId": "routing-holdout-by-purpose",
          "aN": 14,
          "bN": 14,
          "aContext": "",
          "bContext": "Claude Code · effort low",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.6853,
            0.9873
          ],
          "bRange": [
            0.6853,
            0.9873
          ]
        },
        {
          "metric": "Exact rate on unseen decisions, by decision type: Context shape",
          "aValue": 0.5714,
          "bValue": 0.5714,
          "unit": "rate",
          "aDisplay": "57% (8/14)",
          "bDisplay": "57% (8/14)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Jev 1.13 33% to 79%; Claude Sonnet 5.5 33% to 79%), so this sample cannot separate them.",
          "studySlug": "routing-holdout",
          "n": 14,
          "chartId": "routing-holdout-by-purpose",
          "aN": 14,
          "bN": 14,
          "aContext": "",
          "bContext": "Claude Code · effort low",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.3259,
            0.7862
          ],
          "bRange": [
            0.3259,
            0.7862
          ]
        },
        {
          "metric": "Tuned case set vs unseen holdout: exact rate per router (Tuned set (routing-jev-vs-llm))",
          "aValue": 0.9024,
          "bValue": 0.939,
          "unit": "rate",
          "aDisplay": "90% (74/82)",
          "bDisplay": "94% (77/82)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Jev 1.13 82% to 95%; Claude Sonnet 5.5 87% to 97%), so this sample cannot separate them.",
          "studySlug": "routing-holdout",
          "n": 82,
          "chartId": "routing-holdout-tuned-vs-unseen",
          "aN": 82,
          "bN": 82,
          "aContext": "",
          "bContext": "Claude Code · effort low",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.8191,
            0.9497
          ],
          "bRange": [
            0.8651,
            0.9737
          ]
        },
        {
          "metric": "Tuned case set vs unseen holdout: exact rate per router (Unseen holdout)",
          "aValue": 0.8214,
          "bValue": 0.875,
          "unit": "rate",
          "aDisplay": "82% (46/56)",
          "bDisplay": "88% (49/56)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Jev 1.13 70% to 90%; Claude Sonnet 5.5 76% to 94%), so this sample cannot separate them.",
          "studySlug": "routing-holdout",
          "n": 56,
          "chartId": "routing-holdout-tuned-vs-unseen",
          "aN": 56,
          "bN": 56,
          "aContext": "",
          "bContext": "Claude Code · effort low",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.7016,
            0.9
          ],
          "bRange": [
            0.7637,
            0.9381
          ]
        },
        {
          "metric": "Time per routing decision, by route (Wall time)",
          "aValue": 0.139,
          "bValue": 2.359,
          "unit": "seconds",
          "aDisplay": "0.14 s",
          "bDisplay": "2.36 s",
          "winner": "a",
          "basis": "Claude Sonnet 5.5’s median is above Jev 1.13’s 95th percentile (p50–p95 bands: Jev 1.13 0.14 s to 0.19 s; Claude Sonnet 5.5 2.36 s to 3.66 s); not a confidence interval.",
          "studySlug": "routing-holdout",
          "chartId": "routing-holdout-latency",
          "aN": 168,
          "bN": 56,
          "aContext": "",
          "bContext": "Claude Code · effort low",
          "rangeKind": "range",
          "spanKind": "p50-p95",
          "aRange": [
            0.139,
            0.192
          ],
          "bRange": [
            2.359,
            3.657
          ]
        },
        {
          "metric": "Cost per 1,000 unseen routing decisions",
          "aValue": 0.03065,
          "bValue": 7.244,
          "unit": "usd",
          "aDisplay": "$0.031",
          "bDisplay": "$7.24",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.031 vs $7.24, 236x) is not tested against run-to-run variation.",
          "studySlug": "routing-holdout",
          "chartId": "routing-holdout-cost-per-1000",
          "aN": 168,
          "bN": 56,
          "aContext": "",
          "bContext": "Claude Code · effort low",
          "calculation": true
        }
      ]
    },
    {
      "slug": "gpt-6-1-sol-codex-cli-vs-gpt-6-1-sol-openai-api",
      "a": "gpt-6-1-sol-codex-cli",
      "b": "gpt-6-1-sol-openai-api",
      "title": "GPT-6.1 Sol (Codex CLI) vs GPT-6.1 Sol (OpenAI API)",
      "seoTitle": "GPT-6.1 Sol (Codex CLI) vs GPT-6.1 Sol (OpenAI API)",
      "description": "GPT-6.1 Sol (Codex CLI) vs GPT-6.1 Sol (OpenAI API): 8 measured metrics from one study, with sample sizes, intervals and every failure counted.",
      "verdict": "GPT-6.1 Sol (Codex CLI) and GPT-6.1 Sol (OpenAI API) share 8 measured metrics from one study. GPT-6.1 Sol (OpenAI API) leads on 2 rows: CLI vs API: time for a one-line answer (Total time), 1.52 s vs 4.19 s; CLI vs API: time for a one-line answer (First useful output), 1.34 s vs 3.79 s. On those rows the run ranges do not overlap; only a 95% interval is a confidence interval. The other rows are 6 unclear; each row says why. Every row ran the two sides through different routes (for example Codex CLI vs OpenAI API), so they compare route + model pairs, not models alone; the contexts name the route. Some rows rest on small samples (n = 3 at the smallest).",
      "rows": [
        {
          "metric": "CLI vs API: time for a one-line answer (Total time)",
          "aValue": 4.19,
          "bValue": 1.52,
          "unit": "seconds",
          "aDisplay": "4.19 s",
          "bDisplay": "1.52 s",
          "winner": "b",
          "basis": "The run ranges (fastest to slowest) do not overlap (GPT-6.1 Sol (Codex CLI) 3.81 s to 4.69 s; GPT-6.1 Sol (OpenAI API) 1.35 s to 2.23 s). A range is not a confidence interval. Samples are small (5 runs per side).",
          "studySlug": "cli-model-latency-tokens",
          "n": 5,
          "chartId": "cli-vs-api-exact-reply-latency",
          "aN": 5,
          "bN": 5,
          "aContext": "Codex CLI · effort high · fixed exact reply, 5 runs",
          "bContext": "OpenAI API · effort high · fixed exact reply, 5 runs",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            3.81,
            4.69
          ],
          "bRange": [
            1.35,
            2.23
          ]
        },
        {
          "metric": "CLI vs API: time for a one-line answer (First useful output)",
          "aValue": 3.79,
          "bValue": 1.34,
          "unit": "seconds",
          "aDisplay": "3.79 s",
          "bDisplay": "1.34 s",
          "winner": "b",
          "basis": "The run ranges (fastest to slowest) do not overlap (GPT-6.1 Sol (Codex CLI) 3.37 s to 4.30 s; GPT-6.1 Sol (OpenAI API) 1.26 s to 2.12 s). A range is not a confidence interval. Samples are small (5 runs per side).",
          "studySlug": "cli-model-latency-tokens",
          "n": 5,
          "chartId": "cli-vs-api-exact-reply-latency",
          "aN": 5,
          "bN": 5,
          "aContext": "Codex CLI · effort high · fixed exact reply, 5 runs",
          "bContext": "OpenAI API · effort high · fixed exact reply, 5 runs",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            3.37,
            4.3
          ],
          "bRange": [
            1.26,
            2.12
          ]
        },
        {
          "metric": "CLI vs API: time for a small coding task (Total time)",
          "aValue": 17.85,
          "bValue": 9.56,
          "unit": "seconds",
          "aDisplay": "17.9 s",
          "bDisplay": "9.56 s",
          "winner": "unclear",
          "basis": "Only 3 runs per side; the run ranges (fastest to slowest) do not overlap (GPT-6.1 Sol (Codex CLI) 17.7 s to 22.4 s; GPT-6.1 Sol (OpenAI API) 9.44 s to 10.9 s), but 3 runs cannot show a reliable difference. A range is not a confidence interval.",
          "studySlug": "cli-model-latency-tokens",
          "n": 3,
          "chartId": "cli-vs-api-small-coding-latency",
          "aN": 3,
          "bN": 3,
          "aContext": "Codex CLI · effort high · small coding task, 3 runs",
          "bContext": "OpenAI API · effort high · small coding task, 3 runs",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            17.68,
            22.42
          ],
          "bRange": [
            9.44,
            10.94
          ]
        },
        {
          "metric": "CLI vs API: time for a small coding task (First useful output)",
          "aValue": 17.27,
          "bValue": 5.31,
          "unit": "seconds",
          "aDisplay": "17.3 s",
          "bDisplay": "5.31 s",
          "winner": "unclear",
          "basis": "Only 3 runs per side; the run ranges (fastest to slowest) do not overlap (GPT-6.1 Sol (Codex CLI) 17.1 s to 21.9 s; GPT-6.1 Sol (OpenAI API) 4.99 s to 6.42 s), but 3 runs cannot show a reliable difference. A range is not a confidence interval.",
          "studySlug": "cli-model-latency-tokens",
          "n": 3,
          "chartId": "cli-vs-api-small-coding-latency",
          "aN": 3,
          "bN": 3,
          "aContext": "Codex CLI · effort high · small coding task, 3 runs",
          "bContext": "OpenAI API · effort high · small coding task, 3 runs",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            17.13,
            21.86
          ],
          "bRange": [
            4.99,
            6.42
          ]
        },
        {
          "metric": "Hidden prompt: input tokens for the same one-line request",
          "aValue": 19555,
          "bValue": 17,
          "unit": "tokens",
          "aDisplay": "19,555",
          "bDisplay": "17",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "cli-model-latency-tokens",
          "n": 5,
          "chartId": "cli-vs-api-prompt-overhead",
          "aN": 5,
          "bN": 5,
          "aContext": "Codex CLI · effort high · short fixed tasks",
          "bContext": "OpenAI API · effort high · short fixed tasks"
        },
        {
          "metric": "Repairing a scheduler: Claude Code vs Codex vs API (Total time)",
          "aValue": 61.16,
          "bValue": 17.32,
          "unit": "seconds",
          "aDisplay": "61.2 s",
          "bDisplay": "17.3 s",
          "winner": "unclear",
          "basis": "Only 3 runs per side; the run ranges (fastest to slowest) do not overlap (GPT-6.1 Sol (Codex CLI) 59.9 s to 69.5 s; GPT-6.1 Sol (OpenAI API) 16.3 s to 18.6 s), but 3 runs cannot show a reliable difference. A range is not a confidence interval.",
          "studySlug": "cli-model-latency-tokens",
          "n": 3,
          "chartId": "scheduler-repair-claude-vs-codex",
          "aN": 3,
          "bN": 3,
          "aContext": "Codex CLI · effort medium · scheduler repair, 296 checks, 3 runs",
          "bContext": "OpenAI API · effort medium · scheduler repair, 296 checks, 3 runs",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            59.9,
            69.51
          ],
          "bRange": [
            16.28,
            18.61
          ]
        },
        {
          "metric": "Repairing a scheduler: Claude Code vs Codex vs API (First useful output)",
          "aValue": 15.56,
          "bValue": 7.46,
          "unit": "seconds",
          "aDisplay": "15.6 s",
          "bDisplay": "7.46 s",
          "winner": "unclear",
          "basis": "Only 3 runs per side; the run ranges (fastest to slowest) do not overlap (GPT-6.1 Sol (Codex CLI) 13.7 s to 23.0 s; GPT-6.1 Sol (OpenAI API) 6.68 s to 9.05 s), but 3 runs cannot show a reliable difference. A range is not a confidence interval.",
          "studySlug": "cli-model-latency-tokens",
          "n": 3,
          "chartId": "scheduler-repair-claude-vs-codex",
          "aN": 3,
          "bN": 3,
          "aContext": "Codex CLI · effort medium · scheduler repair, 296 checks, 3 runs",
          "bContext": "OpenAI API · effort medium · scheduler repair, 296 checks, 3 runs",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            13.65,
            23.04
          ],
          "bRange": [
            6.68,
            9.05
          ]
        },
        {
          "metric": "Output tokens to repair the scheduler (Output tokens)",
          "aValue": 1181,
          "bValue": 1313,
          "unit": "tokens",
          "aDisplay": "1,181",
          "bDisplay": "1,313",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "cli-model-latency-tokens",
          "n": 3,
          "chartId": "scheduler-repair-output-tokens",
          "aN": 3,
          "bN": 3,
          "aContext": "Codex CLI · effort medium · scheduler repair, 296 checks, 3 runs",
          "bContext": "OpenAI API · effort medium · scheduler repair, 296 checks, 3 runs"
        }
      ]
    },
    {
      "slug": "claude-opus-5-5-vs-gpt-6-1-sol-codex-cli",
      "a": "claude-opus-5-5",
      "b": "gpt-6-1-sol-codex-cli",
      "title": "Claude Opus 5.5 vs GPT-6.1 Sol (Codex CLI)",
      "seoTitle": "Claude Opus 5.5 vs GPT-6.1 Sol (Codex CLI): benchmarks",
      "description": "Claude Opus 5.5 vs GPT-6.1 Sol (Codex CLI): 31 measured metrics from 6 studies (Pass rate on five validated tasks; more), with sample sizes and intervals.",
      "verdict": "Claude Opus 5.5 and GPT-6.1 Sol (Codex CLI) share 31 measured metrics and 19 list-price calculations from 7 studies. No row separates them: every interval or run range overlaps, too few runs were recorded, no interval was recorded, or more is not better for that metric. The rows are 12 ties and 38 unclear; each row says why. Calculation rows are derived from list prices and recorded counts; they are not bills or runs. Every row ran the two sides through different routes (for example Claude Code vs Codex CLI), so they compare route + model pairs, not models alone; the contexts name the route. Some rows rest on small samples (n = 3 at the smallest).",
      "rows": [
        {
          "metric": "Pass rate on five validated tasks",
          "aValue": 1,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "100% (15/15)",
          "bDisplay": "100% (15/15)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Opus 5.5 80% to 100%; GPT-6.1 Sol (Codex CLI) 80% to 100%), so this sample cannot separate them.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-pass-rate",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code · effort high · five short validated tasks",
          "bContext": "Codex CLI · effort high · five short validated tasks",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.7961,
            1
          ],
          "bRange": [
            0.7961,
            1
          ]
        },
        {
          "metric": "Total time per call",
          "aValue": 2.71,
          "bValue": 5.6,
          "unit": "seconds",
          "aDisplay": "2.71 s",
          "bDisplay": "5.60 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Opus 5.5 2.45 s to 11.8 s; GPT-6.1 Sol (Codex CLI) 4.05 s to 19.5 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-total-latency",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code · effort high · five short validated tasks",
          "bContext": "Codex CLI · effort high · five short validated tasks",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            2.45,
            11.78
          ],
          "bRange": [
            4.05,
            19.52
          ]
        },
        {
          "metric": "Time to first useful output",
          "aValue": 2.04,
          "bValue": 5.32,
          "unit": "seconds",
          "aDisplay": "2.04 s",
          "bDisplay": "5.32 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Opus 5.5 1.40 s to 9.94 s; GPT-6.1 Sol (Codex CLI) 3.64 s to 16.4 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-first-useful-latency",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code · effort high · five short validated tasks",
          "bContext": "Codex CLI · effort high · five short validated tasks",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            1.4,
            9.94
          ],
          "bRange": [
            3.64,
            16.37
          ]
        },
        {
          "metric": "Input tokens per call: what the CLI sends (Cache read)",
          "aValue": 1463,
          "bValue": 6716,
          "unit": "tokens",
          "aDisplay": "1,463",
          "bDisplay": "6,716",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-input-tokens",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code · effort high · five short validated tasks",
          "bContext": "Codex CLI · effort high · five short validated tasks"
        },
        {
          "metric": "Input tokens per call: what the CLI sends (Other input)",
          "aValue": 619,
          "bValue": 5406,
          "unit": "tokens",
          "aDisplay": "619",
          "bDisplay": "5,406",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-input-tokens",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code · effort high · five short validated tasks",
          "bContext": "Codex CLI · effort high · five short validated tasks"
        },
        {
          "metric": "Output tokens per call (Output tokens)",
          "aValue": 78,
          "bValue": 42,
          "unit": "tokens",
          "aDisplay": "78",
          "bDisplay": "42",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-output-tokens",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code · effort high · five short validated tasks",
          "bContext": "Codex CLI · effort high · five short validated tasks"
        },
        {
          "metric": "List-price cost per call (calculation)",
          "aValue": 0.00694,
          "bValue": 0.01047,
          "unit": "usd",
          "aDisplay": "$0.0069",
          "bDisplay": "$0.010",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Opus 5.5 $0.0059 to $0.027; GPT-6.1 Sol (Codex CLI) $0.0066 to $0.028); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-list-price-per-call",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code · effort high · five short validated tasks",
          "bContext": "Codex CLI · effort high · five short validated tasks",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            0.00592,
            0.02708
          ],
          "bRange": [
            0.0066,
            0.02812
          ],
          "calculation": true
        },
        {
          "metric": "List-price cost per passing answer (calculation)",
          "aValue": 0.01049,
          "bValue": 0.01322,
          "unit": "usd",
          "aDisplay": "$0.010",
          "bDisplay": "$0.013",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.010 vs $0.013) is not tested against run-to-run variation.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-cost-per-pass",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code · effort high · five short validated tasks",
          "bContext": "Codex CLI · effort high · five short validated tasks",
          "calculation": true
        },
        {
          "metric": "Pass rate on eight hard tasks (Strict pass)",
          "aValue": 1,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "100% (24/24)",
          "bDisplay": "100% (16/16)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Opus 5.5 86% to 100%; GPT-6.1 Sol (Codex CLI) 81% to 100%), so this sample cannot separate them.",
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-pass-rate",
          "aN": 24,
          "bN": 16,
          "aContext": "Claude Code · effort high · eight hard validated tasks",
          "bContext": "Codex CLI · effort high · eight hard validated tasks",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.862,
            1
          ],
          "bRange": [
            0.8064,
            1
          ]
        },
        {
          "metric": "Pass rate on eight hard tasks (Lenient (format misses counted))",
          "aValue": 1,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "100% (24/24)",
          "bDisplay": "100% (16/16)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Opus 5.5 86% to 100%; GPT-6.1 Sol (Codex CLI) 81% to 100%), so this sample cannot separate them.",
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-pass-rate",
          "aN": 24,
          "bN": 16,
          "aContext": "Claude Code · effort high · eight hard validated tasks",
          "bContext": "Codex CLI · effort high · eight hard validated tasks",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.862,
            1
          ],
          "bRange": [
            0.8064,
            1
          ]
        },
        {
          "metric": "Total time per call on hard tasks (separate batches)",
          "aValue": 11.03,
          "bValue": 18.12,
          "unit": "seconds",
          "aDisplay": "11.0 s",
          "bDisplay": "18.1 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Opus 5.5 3.63 s to 63.0 s; GPT-6.1 Sol (Codex CLI) 11.7 s to 92.2 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-total-latency",
          "aN": 24,
          "bN": 16,
          "aContext": "Claude Code · effort high · eight hard validated tasks",
          "bContext": "Codex CLI · effort high · eight hard validated tasks",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            3.63,
            63
          ],
          "bRange": [
            11.67,
            92.21
          ]
        },
        {
          "metric": "Time to first useful output on hard tasks",
          "aValue": 7.13,
          "bValue": 12.69,
          "unit": "seconds",
          "aDisplay": "7.13 s",
          "bDisplay": "12.7 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Opus 5.5 2.15 s to 56.2 s; GPT-6.1 Sol (Codex CLI) 8.93 s to 75.9 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-first-useful-latency",
          "aN": 24,
          "bN": 16,
          "aContext": "Claude Code · effort high · eight hard validated tasks",
          "bContext": "Codex CLI · effort high · eight hard validated tasks",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            2.15,
            56.23
          ],
          "bRange": [
            8.93,
            75.91
          ]
        },
        {
          "metric": "Output tokens per call on hard tasks (Output tokens)",
          "aValue": 1052,
          "bValue": 436,
          "unit": "tokens",
          "aDisplay": "1,052",
          "bDisplay": "436",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-output-tokens",
          "aN": 24,
          "bN": 16,
          "aContext": "Claude Code · effort high · eight hard validated tasks",
          "bContext": "Codex CLI · effort high · eight hard validated tasks"
        },
        {
          "metric": "List-price cost per strict pass on hard tasks (calculation)",
          "aValue": 0.03337,
          "bValue": 0.01514,
          "unit": "usd",
          "aDisplay": "$0.033",
          "bDisplay": "$0.015",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.033 vs $0.015, 2.2x) is not tested against run-to-run variation.",
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-cost-per-pass",
          "aN": 24,
          "bN": 16,
          "aContext": "Claude Code · effort high · eight hard validated tasks",
          "bContext": "Codex CLI · effort high · eight hard validated tasks",
          "calculation": true
        },
        {
          "metric": "Coding sessions that passed every hidden check",
          "aValue": 1,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "100% (12/12)",
          "bDisplay": "100% (12/12)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Opus 5.5 76% to 100%; GPT-6.1 Sol (Codex CLI) 76% to 100%), so this sample cannot separate them.",
          "studySlug": "coding-agents-head-to-head",
          "n": 12,
          "chartId": "coding-agents-pass-rate",
          "aN": 12,
          "bN": 12,
          "aContext": "Claude Code · six small repository tasks with hidden tests",
          "bContext": "Codex CLI · effort medium · tester’s AGENTS.md · six small repository tasks with hidden tests",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.7575,
            1
          ],
          "bRange": [
            0.7575,
            1
          ]
        },
        {
          "metric": "Time per coding session",
          "aValue": 56.9,
          "bValue": 113.4,
          "unit": "seconds",
          "aDisplay": "56.9 s",
          "bDisplay": "113.4 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Opus 5.5 29.8 s to 185.8 s; GPT-6.1 Sol (Codex CLI) 78.5 s to 221.9 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "coding-agents-head-to-head",
          "n": 12,
          "chartId": "coding-agents-wall-time",
          "aN": 12,
          "bN": 12,
          "aContext": "Claude Code · six small repository tasks with hidden tests",
          "bContext": "Codex CLI · effort medium · tester’s AGENTS.md · six small repository tasks with hidden tests",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            29.8,
            185.8
          ],
          "bRange": [
            78.5,
            221.9
          ]
        },
        {
          "metric": "Tool calls per coding session",
          "aValue": 7.5,
          "bValue": 12.5,
          "unit": "calls",
          "aDisplay": "7.5",
          "bDisplay": "12.5",
          "winner": "unclear",
          "basis": "More or fewer calls is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "coding-agents-head-to-head",
          "n": 12,
          "chartId": "coding-agents-tool-calls",
          "aN": 12,
          "bN": 12,
          "aContext": "Claude Code · six small repository tasks with hidden tests",
          "bContext": "Codex CLI · effort medium · tester’s AGENTS.md · six small repository tasks with hidden tests",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            5,
            14
          ],
          "bRange": [
            8,
            18
          ]
        },
        {
          "metric": "List-price cost per passing coding session (calculation)",
          "aValue": 0.2229,
          "bValue": 0.0978,
          "unit": "usd",
          "aDisplay": "$0.22",
          "bDisplay": "$0.098",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.22 vs $0.098, 2.3x) is not tested against run-to-run variation.",
          "studySlug": "coding-agents-head-to-head",
          "n": 12,
          "chartId": "coding-agents-cost-per-pass",
          "aN": 12,
          "bN": 12,
          "aContext": "Claude Code · six small repository tasks with hidden tests",
          "bContext": "Codex CLI · effort medium · tester’s AGENTS.md · six small repository tasks with hidden tests",
          "calculation": true
        },
        {
          "metric": "Strict pass rate by effort on eight hard tasks",
          "aValue": 1,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "100% (16/16)",
          "bDisplay": "100% (16/16)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Opus 5.5 81% to 100%; GPT-6.1 Sol (Codex CLI) 81% to 100%), so this sample cannot separate them.",
          "studySlug": "effort-ladder",
          "n": 16,
          "chartId": "effort-ladder-pass-rate",
          "aN": 16,
          "bN": 16,
          "aContext": "Claude Code · effort medium · eight hard validated tasks, effort ladder",
          "bContext": "Codex CLI · effort medium · eight hard validated tasks, effort ladder",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.8064,
            1
          ],
          "bRange": [
            0.8064,
            1
          ]
        },
        {
          "metric": "Total time per call by effort on hard tasks",
          "aValue": 9.72,
          "bValue": 13.11,
          "unit": "seconds",
          "aDisplay": "9.72 s",
          "bDisplay": "13.1 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Opus 5.5 4.78 s to 31.4 s; GPT-6.1 Sol (Codex CLI) 8.54 s to 61.6 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "effort-ladder",
          "n": 16,
          "chartId": "effort-ladder-total-latency",
          "aN": 16,
          "bN": 16,
          "aContext": "Claude Code · effort medium · eight hard validated tasks, effort ladder",
          "bContext": "Codex CLI · effort medium · eight hard validated tasks, effort ladder",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            4.78,
            31.36
          ],
          "bRange": [
            8.54,
            61.6
          ]
        },
        {
          "metric": "Output tokens per call by effort on hard tasks (Output tokens)",
          "aValue": 853,
          "bValue": 335,
          "unit": "tokens",
          "aDisplay": "853",
          "bDisplay": "335",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "effort-ladder",
          "n": 16,
          "chartId": "effort-ladder-output-tokens",
          "aN": 16,
          "bN": 16,
          "aContext": "Claude Code · effort medium · eight hard validated tasks, effort ladder",
          "bContext": "Codex CLI · effort medium · eight hard validated tasks, effort ladder"
        },
        {
          "metric": "List-price cost per strict pass by effort (calculation)",
          "aValue": 0.02947,
          "bValue": 0.02564,
          "unit": "usd",
          "aDisplay": "$0.029",
          "bDisplay": "$0.026",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.029 vs $0.026) is not tested against run-to-run variation.",
          "studySlug": "effort-ladder",
          "n": 16,
          "chartId": "effort-ladder-cost-per-pass",
          "aN": 16,
          "bN": 16,
          "aContext": "Claude Code · effort medium · eight hard validated tasks, effort ladder",
          "bContext": "Codex CLI · effort medium · eight hard validated tasks, effort ladder",
          "calculation": true
        },
        {
          "metric": "Reasoning share of output tokens per call on hard tasks (calculation)",
          "aValue": 54.43,
          "bValue": 57.01,
          "unit": "percent",
          "aDisplay": "54.4%",
          "bDisplay": "57%",
          "winner": "unclear",
          "basis": "More or fewer percent is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-share",
          "aN": 24,
          "bN": 16,
          "aContext": "Claude Code · effort high",
          "bContext": "Codex CLI · effort high",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            36.14,
            96.23
          ],
          "bRange": [
            29.19,
            90.8
          ],
          "calculation": true
        },
        {
          "metric": "List-price cost per call: reasoning, remaining output and input (calculation) (Reasoning (output tokens))",
          "aValue": 0.017969,
          "bValue": 0.004114,
          "unit": "usd",
          "aDisplay": "$0.018",
          "bDisplay": "$0.0041",
          "winner": "unclear",
          "basis": "More or fewer usd is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-cost-per-call",
          "aN": 24,
          "bN": 16,
          "aContext": "Claude Code · effort high",
          "bContext": "Codex CLI · effort high",
          "calculation": true
        },
        {
          "metric": "List-price cost per call: reasoning, remaining output and input (calculation) (Remaining output (visible-answer estimate))",
          "aValue": 0.00799,
          "bValue": 0.002905,
          "unit": "usd",
          "aDisplay": "$0.0080",
          "bDisplay": "$0.0029",
          "winner": "unclear",
          "basis": "More or fewer usd is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-cost-per-call",
          "aN": 24,
          "bN": 16,
          "aContext": "Claude Code · effort high",
          "bContext": "Codex CLI · effort high",
          "calculation": true
        },
        {
          "metric": "List-price cost per call: reasoning, remaining output and input (calculation) (Input (prompt, cache priced))",
          "aValue": 0.007407,
          "bValue": 0.008117,
          "unit": "usd",
          "aDisplay": "$0.0074",
          "bDisplay": "$0.0081",
          "winner": "unclear",
          "basis": "More or fewer usd is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-cost-per-call",
          "aN": 24,
          "bN": 16,
          "aContext": "Claude Code · effort high",
          "bContext": "Codex CLI · effort high",
          "calculation": true
        },
        {
          "metric": "Reasoning cost per strict pass by effort, with the total (calculation) (Reasoning cost per strict pass)",
          "aValue": 0.01344,
          "bValue": 0.002273,
          "unit": "usd",
          "aDisplay": "$0.013",
          "bDisplay": "$0.0023",
          "winner": "unclear",
          "basis": "More or fewer usd is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "n": 16,
          "chartId": "thinking-bill-by-effort",
          "aN": 16,
          "bN": 16,
          "aContext": "Claude Code · effort medium",
          "bContext": "Codex CLI · effort medium",
          "calculation": true
        },
        {
          "metric": "Reasoning cost per strict pass by effort, with the total (calculation) (Total cost per strict pass)",
          "aValue": 0.029475,
          "bValue": 0.025637,
          "unit": "usd",
          "aDisplay": "$0.029",
          "bDisplay": "$0.026",
          "winner": "unclear",
          "basis": "More or fewer usd is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "n": 16,
          "chartId": "thinking-bill-by-effort",
          "aN": 16,
          "bN": 16,
          "aContext": "Claude Code · effort medium",
          "bContext": "Codex CLI · effort medium",
          "calculation": true
        },
        {
          "metric": "Reasoning share on short tasks vs hard tasks (calculation) (Eight hard tasks)",
          "aValue": 54.43,
          "bValue": 57.01,
          "unit": "percent",
          "aDisplay": "54.4%",
          "bDisplay": "57%",
          "winner": "unclear",
          "basis": "More or fewer percent is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-short-vs-hard",
          "aN": 24,
          "bN": 16,
          "aContext": "Claude Code · effort high",
          "bContext": "Codex CLI · effort high",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            36.14,
            96.23
          ],
          "bRange": [
            29.19,
            90.8
          ],
          "calculation": true
        },
        {
          "metric": "Reasoning share on short tasks vs hard tasks (calculation) (Five short tasks)",
          "aValue": 43.59,
          "bValue": 58.06,
          "unit": "percent",
          "aDisplay": "43.6%",
          "bDisplay": "58.1%",
          "winner": "unclear",
          "basis": "More or fewer percent is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "n": 15,
          "chartId": "thinking-bill-short-vs-hard",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code · effort high",
          "bContext": "Codex CLI · effort high",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            0,
            93.33
          ],
          "bRange": [
            0,
            75.76
          ],
          "calculation": true
        },
        {
          "metric": "Time to first text: a 250-line answer, six models",
          "aValue": 1.97,
          "bValue": 3.52,
          "unit": "seconds",
          "aDisplay": "1.97 s",
          "bDisplay": "3.52 s",
          "winner": "unclear",
          "basis": "Only 4 runs per side; the run ranges (fastest to slowest) do not overlap (Claude Opus 5.5 1.70 s to 2.35 s; GPT-6.1 Sol (Codex CLI) 2.75 s to 4.42 s), but 4 runs cannot show a reliable difference. A range is not a confidence interval.",
          "studySlug": "llm-speed-anatomy",
          "n": 4,
          "chartId": "speed-anatomy-first-text",
          "aN": 4,
          "bN": 4,
          "aContext": "Claude Code",
          "bContext": "Codex CLI · effort low",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            1.7,
            2.35
          ],
          "bRange": [
            2.75,
            4.42
          ]
        },
        {
          "metric": "Output speed after the first text: visible tokens per second (calculation)",
          "aValue": 155.5,
          "bValue": 79.6,
          "unit": "tokens",
          "aDisplay": "156",
          "bDisplay": "80",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "llm-speed-anatomy",
          "n": 4,
          "chartId": "speed-anatomy-output-speed",
          "aN": 4,
          "bN": 4,
          "aContext": "Claude Code",
          "bContext": "Codex CLI · effort low",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            154.6,
            156.4
          ],
          "bRange": [
            71.6,
            80.5
          ],
          "calculation": true
        },
        {
          "metric": "Output speed in characters per second after the first text (calculation)",
          "aValue": 347,
          "bValue": 323,
          "unit": "count",
          "aDisplay": "347",
          "bDisplay": "323",
          "winner": "unclear",
          "basis": "More or fewer count is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "llm-speed-anatomy",
          "n": 4,
          "chartId": "speed-anatomy-chars-per-second",
          "aN": 4,
          "bN": 4,
          "aContext": "Claude Code",
          "bContext": "Codex CLI · effort low",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            345,
            349
          ],
          "bRange": [
            291,
            327
          ],
          "calculation": true
        },
        {
          "metric": "Time to first text as the prompt grows: 1k",
          "aValue": 1.51,
          "bValue": 3.36,
          "unit": "seconds",
          "aDisplay": "1.51 s",
          "bDisplay": "3.36 s",
          "winner": "unclear",
          "basis": "Only 3 runs per side; the run ranges (fastest to slowest) do not overlap (Claude Opus 5.5 1.46 s to 2.01 s; GPT-6.1 Sol (Codex CLI) 3.36 s to 4.75 s), but 3 runs cannot show a reliable difference. A range is not a confidence interval.",
          "studySlug": "llm-speed-anatomy",
          "n": 3,
          "chartId": "speed-anatomy-prompt-size",
          "aN": 3,
          "bN": 3,
          "aContext": "Claude Code",
          "bContext": "Codex CLI · effort low",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            1.46,
            2.01
          ],
          "bRange": [
            3.36,
            4.75
          ],
          "calculation": true
        },
        {
          "metric": "Time to first text as the prompt grows: 16k",
          "aValue": 1.74,
          "bValue": 4.02,
          "unit": "seconds",
          "aDisplay": "1.74 s",
          "bDisplay": "4.02 s",
          "winner": "unclear",
          "basis": "Only 3 runs per side; the run ranges (fastest to slowest) do not overlap (Claude Opus 5.5 1.70 s to 2.97 s; GPT-6.1 Sol (Codex CLI) 3.30 s to 4.28 s), but 3 runs cannot show a reliable difference. A range is not a confidence interval.",
          "studySlug": "llm-speed-anatomy",
          "n": 3,
          "chartId": "speed-anatomy-prompt-size",
          "aN": 3,
          "bN": 3,
          "aContext": "Claude Code",
          "bContext": "Codex CLI · effort low",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            1.7,
            2.97
          ],
          "bRange": [
            3.3,
            4.28
          ],
          "calculation": true
        },
        {
          "metric": "Time to first text as the prompt grows: 64k",
          "aValue": 1.79,
          "bValue": 3.93,
          "unit": "seconds",
          "aDisplay": "1.79 s",
          "bDisplay": "3.93 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Opus 5.5 1.72 s to 3.72 s; GPT-6.1 Sol (Codex CLI) 3.42 s to 4.38 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "llm-speed-anatomy",
          "n": 3,
          "chartId": "speed-anatomy-prompt-size",
          "aN": 3,
          "bN": 3,
          "aContext": "Claude Code",
          "bContext": "Codex CLI · effort low",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            1.72,
            3.72
          ],
          "bRange": [
            3.42,
            4.38
          ],
          "calculation": true
        },
        {
          "metric": "Total time per call by prompt size (1k prompt)",
          "aValue": 1.83,
          "bValue": 3.43,
          "unit": "seconds",
          "aDisplay": "1.83 s",
          "bDisplay": "3.43 s",
          "winner": "unclear",
          "basis": "Only 3 runs per side; the run ranges (fastest to slowest) do not overlap (Claude Opus 5.5 1.82 s to 2.41 s; GPT-6.1 Sol (Codex CLI) 3.43 s to 4.92 s), but 3 runs cannot show a reliable difference. A range is not a confidence interval.",
          "studySlug": "llm-speed-anatomy",
          "n": 3,
          "chartId": "speed-anatomy-total-by-size",
          "aN": 3,
          "bN": 3,
          "aContext": "Claude Code",
          "bContext": "Codex CLI · effort low",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            1.82,
            2.41
          ],
          "bRange": [
            3.43,
            4.92
          ]
        },
        {
          "metric": "Total time per call by prompt size (16k prompt)",
          "aValue": 2.36,
          "bValue": 4.14,
          "unit": "seconds",
          "aDisplay": "2.36 s",
          "bDisplay": "4.14 s",
          "winner": "unclear",
          "basis": "Only 3 runs per side; the run ranges (fastest to slowest) do not overlap (Claude Opus 5.5 2.11 s to 3.40 s; GPT-6.1 Sol (Codex CLI) 3.96 s to 4.68 s), but 3 runs cannot show a reliable difference. A range is not a confidence interval.",
          "studySlug": "llm-speed-anatomy",
          "n": 3,
          "chartId": "speed-anatomy-total-by-size",
          "aN": 3,
          "bN": 3,
          "aContext": "Claude Code",
          "bContext": "Codex CLI · effort low",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            2.11,
            3.4
          ],
          "bRange": [
            3.96,
            4.68
          ]
        },
        {
          "metric": "Total time per call by prompt size (64k prompt)",
          "aValue": 2.35,
          "bValue": 3.96,
          "unit": "seconds",
          "aDisplay": "2.35 s",
          "bDisplay": "3.96 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Opus 5.5 2.26 s to 4.29 s; GPT-6.1 Sol (Codex CLI) 3.47 s to 4.44 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "llm-speed-anatomy",
          "n": 3,
          "chartId": "speed-anatomy-total-by-size",
          "aN": 3,
          "bN": 3,
          "aContext": "Claude Code",
          "bContext": "Codex CLI · effort low",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            2.26,
            4.29
          ],
          "bRange": [
            3.47,
            4.44
          ]
        },
        {
          "metric": "Exact lookup answers at the 1k, 16k and 64k prompt-size targets",
          "aValue": 0.5556,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "56% (5/9)",
          "bDisplay": "100% (9/9)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Opus 5.5 27% to 81%; GPT-6.1 Sol (Codex CLI) 70% to 100%), so this sample cannot separate them.",
          "studySlug": "llm-speed-anatomy",
          "n": 9,
          "chartId": "speed-anatomy-lookup-correct",
          "aN": 9,
          "bN": 9,
          "aContext": "Claude Code",
          "bContext": "Codex CLI · effort low",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.2667,
            0.8112
          ],
          "bRange": [
            0.7009,
            1
          ]
        },
        {
          "metric": "Pass rate on 4 harder tasks (Strict pass)",
          "aValue": 0.4167,
          "bValue": 0.6875,
          "unit": "rate",
          "aDisplay": "42% (5/12)",
          "bDisplay": "69% (11/16)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Opus 5.5 19% to 68%; GPT-6.1 Sol (Codex CLI) 44% to 86%), so this sample cannot separate them.",
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-pass-rate",
          "aN": 12,
          "bN": 16,
          "aContext": "Claude Code",
          "bContext": "Codex CLI · effort medium",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.1933,
            0.6805
          ],
          "bRange": [
            0.444,
            0.8584
          ]
        },
        {
          "metric": "Pass rate on 4 harder tasks (Lenient (format misses counted))",
          "aValue": 0.5,
          "bValue": 0.6875,
          "unit": "rate",
          "aDisplay": "50% (6/12)",
          "bDisplay": "69% (11/16)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Opus 5.5 25% to 75%; GPT-6.1 Sol (Codex CLI) 44% to 86%), so this sample cannot separate them.",
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-pass-rate",
          "aN": 12,
          "bN": 16,
          "aContext": "Claude Code",
          "bContext": "Codex CLI · effort medium",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.2538,
            0.7462
          ],
          "bRange": [
            0.444,
            0.8584
          ]
        },
        {
          "metric": "Calls that tried a tool although tools were off",
          "aValue": 0.4167,
          "bValue": 0,
          "unit": "rate",
          "aDisplay": "42% (5/12)",
          "bDisplay": "0% (0/16)",
          "winner": "unclear",
          "basis": "More or fewer rate is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-tool-attempts",
          "aN": 12,
          "bN": 16,
          "aContext": "Claude Code",
          "bContext": "Codex CLI · effort medium",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.1933,
            0.6805
          ],
          "bRange": [
            0,
            0.1936
          ]
        },
        {
          "metric": "Strict pass rate by task: 10x10 nonogram",
          "aValue": 1,
          "bValue": 0.75,
          "unit": "rate",
          "aDisplay": "100% (3/3)",
          "bDisplay": "75% (3/4)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Opus 5.5 44% to 100%; GPT-6.1 Sol (Codex CLI) 30% to 95%), so this sample cannot separate them.",
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-pass-by-task",
          "aN": 3,
          "bN": 4,
          "aContext": "Claude Code",
          "bContext": "Codex CLI · effort medium",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.4385,
            1
          ],
          "bRange": [
            0.3006,
            0.9544
          ]
        },
        {
          "metric": "Strict pass rate by task: Sudoku, 22 givens",
          "aValue": 0,
          "bValue": 0.25,
          "unit": "rate",
          "aDisplay": "0% (0/3)",
          "bDisplay": "25% (1/4)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Opus 5.5 0% to 56%; GPT-6.1 Sol (Codex CLI) 5% to 70%), so this sample cannot separate them.",
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-pass-by-task",
          "aN": 3,
          "bN": 4,
          "aContext": "Claude Code",
          "bContext": "Codex CLI · effort medium",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0,
            0.5615
          ],
          "bRange": [
            0.0456,
            0.6994
          ]
        },
        {
          "metric": "Strict pass rate by task: 6x6 Skyscrapers",
          "aValue": 0.3333,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "33% (1/3)",
          "bDisplay": "100% (4/4)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Opus 5.5 6% to 79%; GPT-6.1 Sol (Codex CLI) 51% to 100%), so this sample cannot separate them.",
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-pass-by-task",
          "aN": 3,
          "bN": 4,
          "aContext": "Claude Code",
          "bContext": "Codex CLI · effort medium",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.0615,
            0.7923
          ],
          "bRange": [
            0.5101,
            1
          ]
        },
        {
          "metric": "Strict pass rate by task: Seeded shuffle output",
          "aValue": 0.3333,
          "bValue": 0.75,
          "unit": "rate",
          "aDisplay": "33% (1/3)",
          "bDisplay": "75% (3/4)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Opus 5.5 6% to 79%; GPT-6.1 Sol (Codex CLI) 30% to 95%), so this sample cannot separate them.",
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-pass-by-task",
          "aN": 3,
          "bN": 4,
          "aContext": "Claude Code",
          "bContext": "Codex CLI · effort medium",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.0615,
            0.7923
          ],
          "bRange": [
            0.3006,
            0.9544
          ]
        },
        {
          "metric": "Total time per call on harder tasks",
          "aValue": 80.34,
          "bValue": 120.24,
          "unit": "seconds",
          "aDisplay": "80.3 s",
          "bDisplay": "120.2 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Opus 5.5 3.82 s to 279.5 s; GPT-6.1 Sol (Codex CLI) 46.2 s to 273.5 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-total-latency",
          "aN": 9,
          "bN": 13,
          "aContext": "Claude Code",
          "bContext": "Codex CLI · effort medium",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            3.82,
            279.5
          ],
          "bRange": [
            46.24,
            273.46
          ]
        },
        {
          "metric": "Output tokens per call on harder tasks (Output tokens)",
          "aValue": 8420,
          "bValue": 4994,
          "unit": "tokens",
          "aDisplay": "8,420",
          "bDisplay": "4,994",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-output-tokens",
          "aN": 9,
          "bN": 13,
          "aContext": "Claude Code",
          "bContext": "Codex CLI · effort medium",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            279,
            40044
          ],
          "bRange": [
            2099,
            13413
          ]
        },
        {
          "metric": "List-price cost per strict pass on harder tasks (calculation)",
          "aValue": 0.59333,
          "bValue": 0.08293,
          "unit": "usd",
          "aDisplay": "$0.59",
          "bDisplay": "$0.083",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.59 vs $0.083, 7.2x) is not tested against run-to-run variation.",
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-cost-per-pass",
          "aN": 12,
          "bN": 16,
          "aContext": "Claude Code",
          "bContext": "Codex CLI · effort medium",
          "calculation": true
        }
      ]
    },
    {
      "slug": "claude-haiku-4-5-vs-gpt-6-1-sol-codex-cli",
      "a": "claude-haiku-4-5",
      "b": "gpt-6-1-sol-codex-cli",
      "title": "Claude Haiku 4.5 vs GPT-6.1 Sol (Codex CLI)",
      "seoTitle": "Claude Haiku 4.5 vs GPT-6.1 Sol (Codex CLI): benchmarks",
      "description": "Claude Haiku 4.5 vs GPT-6.1 Sol (Codex CLI): 40 measured metrics from 6 studies (Pass rate on five validated tasks; more), with sample sizes and intervals.",
      "verdict": "Claude Haiku 4.5 and GPT-6.1 Sol (Codex CLI) share 40 measured metrics and 16 list-price calculations from 7 studies. Claude Haiku 4.5 leads on 2 rows: Same prompt, 10 times: time per call (Exact number), 5.06 s vs 13.4 s; Same prompt, 10 times: time per call (Code fix), 5.95 s vs 11.3 s. GPT-6.1 Sol (Codex CLI) leads on 6 rows: Pass rate on eight hard tasks (Strict pass), 100% (16/16) vs 46% (11/24); Same prompt, 10 times: strict pass rate (Exact number), 100% (10/10) vs 0% (0/10); Same prompt, 10 times: strict pass rate (JSON object), 100% (10/10) vs 10% (1/10); and 3 more. On those rows the 95% intervals and run ranges do not overlap; only a 95% interval is a confidence interval. The other rows are 13 ties and 35 unclear; each row says why. Calculation rows are derived from list prices and recorded counts; they are not bills or runs. Every row ran the two sides through different routes (for example Claude Code vs Codex CLI), so they compare route + model pairs, not models alone; the contexts name the route. Some rows rest on small samples (n = 3 at the smallest).",
      "rows": [
        {
          "metric": "Pass rate on five validated tasks",
          "aValue": 1,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "100% (15/15)",
          "bDisplay": "100% (15/15)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Haiku 4.5 80% to 100%; GPT-6.1 Sol (Codex CLI) 80% to 100%), so this sample cannot separate them.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-pass-rate",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code · five short validated tasks",
          "bContext": "Codex CLI · effort medium · five short validated tasks",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.7961,
            1
          ],
          "bRange": [
            0.7961,
            1
          ]
        },
        {
          "metric": "Total time per call",
          "aValue": 4.43,
          "bValue": 5.65,
          "unit": "seconds",
          "aDisplay": "4.43 s",
          "bDisplay": "5.65 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Haiku 4.5 3.16 s to 23.6 s; GPT-6.1 Sol (Codex CLI) 4.10 s to 25.5 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-total-latency",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code · five short validated tasks",
          "bContext": "Codex CLI · effort medium · five short validated tasks",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            3.16,
            23.57
          ],
          "bRange": [
            4.1,
            25.46
          ]
        },
        {
          "metric": "Time to first useful output",
          "aValue": 3.63,
          "bValue": 5.05,
          "unit": "seconds",
          "aDisplay": "3.63 s",
          "bDisplay": "5.05 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Haiku 4.5 2.78 s to 22.3 s; GPT-6.1 Sol (Codex CLI) 3.36 s to 17.8 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-first-useful-latency",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code · five short validated tasks",
          "bContext": "Codex CLI · effort medium · five short validated tasks",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            2.78,
            22.27
          ],
          "bRange": [
            3.36,
            17.82
          ]
        },
        {
          "metric": "Input tokens per call: what the CLI sends (Cache read)",
          "aValue": 0,
          "bValue": 5180,
          "unit": "tokens",
          "aDisplay": "0",
          "bDisplay": "5,180",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-input-tokens",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code · five short validated tasks",
          "bContext": "Codex CLI · effort medium · five short validated tasks"
        },
        {
          "metric": "Input tokens per call: what the CLI sends (Other input)",
          "aValue": 3790,
          "bValue": 6943,
          "unit": "tokens",
          "aDisplay": "3,790",
          "bDisplay": "6,943",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-input-tokens",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code · five short validated tasks",
          "bContext": "Codex CLI · effort medium · five short validated tasks"
        },
        {
          "metric": "Output tokens per call (Output tokens)",
          "aValue": 367,
          "bValue": 42,
          "unit": "tokens",
          "aDisplay": "367",
          "bDisplay": "42",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-output-tokens",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code · five short validated tasks",
          "bContext": "Codex CLI · effort medium · five short validated tasks"
        },
        {
          "metric": "List-price cost per call (calculation)",
          "aValue": 0.00566,
          "bValue": 0.01018,
          "unit": "usd",
          "aDisplay": "$0.0057",
          "bDisplay": "$0.010",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Haiku 4.5 $0.0051 to $0.018; GPT-6.1 Sol (Codex CLI) $0.0054 to $0.027); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-list-price-per-call",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code · five short validated tasks",
          "bContext": "Codex CLI · effort medium · five short validated tasks",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            0.00513,
            0.01804
          ],
          "bRange": [
            0.0054,
            0.02686
          ],
          "calculation": true
        },
        {
          "metric": "List-price cost per passing answer (calculation)",
          "aValue": 0.00836,
          "bValue": 0.01564,
          "unit": "usd",
          "aDisplay": "$0.0084",
          "bDisplay": "$0.016",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.0084 vs $0.016) is not tested against run-to-run variation.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-cost-per-pass",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code · five short validated tasks",
          "bContext": "Codex CLI · effort medium · five short validated tasks",
          "calculation": true
        },
        {
          "metric": "Pass rate on eight hard tasks (Strict pass)",
          "aValue": 0.4583,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "46% (11/24)",
          "bDisplay": "100% (16/16)",
          "winner": "b",
          "basis": "The 95% intervals do not overlap (Claude Haiku 4.5 28% to 65%; GPT-6.1 Sol (Codex CLI) 81% to 100%).",
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-pass-rate",
          "aN": 24,
          "bN": 16,
          "aContext": "Claude Code · eight hard validated tasks",
          "bContext": "Codex CLI · effort medium · eight hard validated tasks",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.2789,
            0.6493
          ],
          "bRange": [
            0.8064,
            1
          ]
        },
        {
          "metric": "Pass rate on eight hard tasks (Lenient (format misses counted))",
          "aValue": 0.6667,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "67% (16/24)",
          "bDisplay": "100% (16/16)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Haiku 4.5 47% to 82%; GPT-6.1 Sol (Codex CLI) 81% to 100%), so this sample cannot separate them.",
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-pass-rate",
          "aN": 24,
          "bN": 16,
          "aContext": "Claude Code · eight hard validated tasks",
          "bContext": "Codex CLI · effort medium · eight hard validated tasks",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.4671,
            0.8203
          ],
          "bRange": [
            0.8064,
            1
          ]
        },
        {
          "metric": "Total time per call on hard tasks (separate batches)",
          "aValue": 39.01,
          "bValue": 13.11,
          "unit": "seconds",
          "aDisplay": "39.0 s",
          "bDisplay": "13.1 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Haiku 4.5 15.3 s to 75.1 s; GPT-6.1 Sol (Codex CLI) 8.54 s to 61.6 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-total-latency",
          "aN": 24,
          "bN": 16,
          "aContext": "Claude Code · eight hard validated tasks",
          "bContext": "Codex CLI · effort medium · eight hard validated tasks",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            15.27,
            75.13
          ],
          "bRange": [
            8.54,
            61.6
          ]
        },
        {
          "metric": "Time to first useful output on hard tasks",
          "aValue": 35.54,
          "bValue": 10.23,
          "unit": "seconds",
          "aDisplay": "35.5 s",
          "bDisplay": "10.2 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Haiku 4.5 12.9 s to 70.3 s; GPT-6.1 Sol (Codex CLI) 6.09 s to 40.4 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-first-useful-latency",
          "aN": 24,
          "bN": 16,
          "aContext": "Claude Code · eight hard validated tasks",
          "bContext": "Codex CLI · effort medium · eight hard validated tasks",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            12.88,
            70.31
          ],
          "bRange": [
            6.09,
            40.41
          ]
        },
        {
          "metric": "Output tokens per call on hard tasks (Output tokens)",
          "aValue": 5064,
          "bValue": 335,
          "unit": "tokens",
          "aDisplay": "5,064",
          "bDisplay": "335",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-output-tokens",
          "aN": 24,
          "bN": 16,
          "aContext": "Claude Code · eight hard validated tasks",
          "bContext": "Codex CLI · effort medium · eight hard validated tasks"
        },
        {
          "metric": "List-price cost per strict pass on hard tasks (calculation)",
          "aValue": 0.0672,
          "bValue": 0.02564,
          "unit": "usd",
          "aDisplay": "$0.067",
          "bDisplay": "$0.026",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.067 vs $0.026, 2.6x) is not tested against run-to-run variation.",
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-cost-per-pass",
          "aN": 24,
          "bN": 16,
          "aContext": "Claude Code · eight hard validated tasks",
          "bContext": "Codex CLI · effort medium · eight hard validated tasks",
          "calculation": true
        },
        {
          "metric": "Same prompt, 10 times: strict pass rate (Exact number)",
          "aValue": 0,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "0% (0/10)",
          "bDisplay": "100% (10/10)",
          "winner": "b",
          "basis": "The 95% intervals do not overlap (Claude Haiku 4.5 0% to 28%; GPT-6.1 Sol (Codex CLI) 72% to 100%).",
          "studySlug": "caching-consistency",
          "n": 10,
          "chartId": "consistency-pass-rate",
          "aN": 10,
          "bN": 10,
          "aContext": "Claude Code · same prompt repeated 10 times",
          "bContext": "Codex CLI · effort medium · same prompt repeated 10 times",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0,
            0.2775
          ],
          "bRange": [
            0.7225,
            1
          ]
        },
        {
          "metric": "Same prompt, 10 times: strict pass rate (JSON object)",
          "aValue": 0.1,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "10% (1/10)",
          "bDisplay": "100% (10/10)",
          "winner": "b",
          "basis": "The 95% intervals do not overlap (Claude Haiku 4.5 2% to 40%; GPT-6.1 Sol (Codex CLI) 72% to 100%).",
          "studySlug": "caching-consistency",
          "n": 10,
          "chartId": "consistency-pass-rate",
          "aN": 10,
          "bN": 10,
          "aContext": "Claude Code · same prompt repeated 10 times",
          "bContext": "Codex CLI · effort medium · same prompt repeated 10 times",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.0179,
            0.4042
          ],
          "bRange": [
            0.7225,
            1
          ]
        },
        {
          "metric": "Same prompt, 10 times: strict pass rate (Code fix)",
          "aValue": 1,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "100% (10/10)",
          "bDisplay": "100% (10/10)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Haiku 4.5 72% to 100%; GPT-6.1 Sol (Codex CLI) 72% to 100%), so this sample cannot separate them.",
          "studySlug": "caching-consistency",
          "n": 10,
          "chartId": "consistency-pass-rate",
          "aN": 10,
          "bN": 10,
          "aContext": "Claude Code · same prompt repeated 10 times",
          "bContext": "Codex CLI · effort medium · same prompt repeated 10 times",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.7225,
            1
          ],
          "bRange": [
            0.7225,
            1
          ]
        },
        {
          "metric": "Same prompt, 10 times: how many different answers (Exact number)",
          "aValue": 1,
          "bValue": 1,
          "unit": "count",
          "aDisplay": "1",
          "bDisplay": "1",
          "winner": "tie",
          "basis": "Same value. More or fewer is not better by itself for this metric.",
          "studySlug": "caching-consistency",
          "n": 10,
          "chartId": "consistency-distinct-answers",
          "aN": 10,
          "bN": 10,
          "aContext": "Claude Code · same prompt repeated 10 times",
          "bContext": "Codex CLI · effort medium · same prompt repeated 10 times"
        },
        {
          "metric": "Same prompt, 10 times: how many different answers (JSON object)",
          "aValue": 1,
          "bValue": 1,
          "unit": "count",
          "aDisplay": "1",
          "bDisplay": "1",
          "winner": "tie",
          "basis": "Same value. More or fewer is not better by itself for this metric.",
          "studySlug": "caching-consistency",
          "n": 10,
          "chartId": "consistency-distinct-answers",
          "aN": 10,
          "bN": 10,
          "aContext": "Claude Code · same prompt repeated 10 times",
          "bContext": "Codex CLI · effort medium · same prompt repeated 10 times"
        },
        {
          "metric": "Same prompt, 10 times: how many different answers (Code fix)",
          "aValue": 6,
          "bValue": 6,
          "unit": "count",
          "aDisplay": "6",
          "bDisplay": "6",
          "winner": "tie",
          "basis": "Same value. More or fewer is not better by itself for this metric.",
          "studySlug": "caching-consistency",
          "n": 10,
          "chartId": "consistency-distinct-answers",
          "aN": 10,
          "bN": 10,
          "aContext": "Claude Code · same prompt repeated 10 times",
          "bContext": "Codex CLI · effort medium · same prompt repeated 10 times"
        },
        {
          "metric": "Same prompt, 10 times: time per call (Exact number)",
          "aValue": 5.06,
          "bValue": 13.38,
          "unit": "seconds",
          "aDisplay": "5.06 s",
          "bDisplay": "13.4 s",
          "winner": "a",
          "basis": "The run ranges (fastest to slowest) do not overlap (Claude Haiku 4.5 4.42 s to 6.20 s; GPT-6.1 Sol (Codex CLI) 12.3 s to 18.0 s). A range is not a confidence interval.",
          "studySlug": "caching-consistency",
          "n": 10,
          "chartId": "consistency-latency-spread",
          "aN": 10,
          "bN": 10,
          "aContext": "Claude Code · same prompt repeated 10 times",
          "bContext": "Codex CLI · effort medium · same prompt repeated 10 times",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            4.42,
            6.2
          ],
          "bRange": [
            12.29,
            17.97
          ]
        },
        {
          "metric": "Same prompt, 10 times: time per call (JSON object)",
          "aValue": 7.03,
          "bValue": 6.42,
          "unit": "seconds",
          "aDisplay": "7.03 s",
          "bDisplay": "6.42 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Haiku 4.5 5.28 s to 12.3 s; GPT-6.1 Sol (Codex CLI) 5.25 s to 8.26 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "caching-consistency",
          "n": 10,
          "chartId": "consistency-latency-spread",
          "aN": 10,
          "bN": 10,
          "aContext": "Claude Code · same prompt repeated 10 times",
          "bContext": "Codex CLI · effort medium · same prompt repeated 10 times",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            5.28,
            12.27
          ],
          "bRange": [
            5.25,
            8.26
          ]
        },
        {
          "metric": "Same prompt, 10 times: time per call (Code fix)",
          "aValue": 5.95,
          "bValue": 11.29,
          "unit": "seconds",
          "aDisplay": "5.95 s",
          "bDisplay": "11.3 s",
          "winner": "a",
          "basis": "The run ranges (fastest to slowest) do not overlap (Claude Haiku 4.5 4.89 s to 7.33 s; GPT-6.1 Sol (Codex CLI) 9.08 s to 14.8 s). A range is not a confidence interval.",
          "studySlug": "caching-consistency",
          "n": 10,
          "chartId": "consistency-latency-spread",
          "aN": 10,
          "bN": 10,
          "aContext": "Claude Code · same prompt repeated 10 times",
          "bContext": "Codex CLI · effort medium · same prompt repeated 10 times",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            4.89,
            7.33
          ],
          "bRange": [
            9.08,
            14.85
          ]
        },
        {
          "metric": "Does a JSON schema raise the pass rate? Instructions vs schema mode (Strict pass: the whole reply is the right JSON)",
          "aValue": 0,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "0% (0/24)",
          "bDisplay": "100% (12/12)",
          "winner": "b",
          "basis": "The 95% intervals do not overlap (Claude Haiku 4.5 0% to 14%; GPT-6.1 Sol (Codex CLI) 76% to 100%).",
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-pass-rate",
          "aN": 24,
          "bN": 12,
          "aContext": "Claude Code · instructions",
          "bContext": "Codex CLI · effort low · instructions",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0,
            0.138
          ],
          "bRange": [
            0.7575,
            1
          ],
          "calculation": true
        },
        {
          "metric": "Does a JSON schema raise the pass rate? Instructions vs schema mode (Right answer in any format (strict pass or format miss))",
          "aValue": 0.7083,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "71% (17/24)",
          "bDisplay": "100% (12/12)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Haiku 4.5 51% to 85%; GPT-6.1 Sol (Codex CLI) 76% to 100%), so this sample cannot separate them.",
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-pass-rate",
          "aN": 24,
          "bN": 12,
          "aContext": "Claude Code · instructions",
          "bContext": "Codex CLI · effort low · instructions",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.5083,
            0.8509
          ],
          "bRange": [
            0.7575,
            1
          ],
          "calculation": true
        },
        {
          "metric": "What each call produced: strict pass, format miss, wrong values or error (Strict pass)",
          "aValue": 0,
          "bValue": 12,
          "unit": "count",
          "aDisplay": "0",
          "bDisplay": "12",
          "winner": "unclear",
          "basis": "More or fewer count is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-outcomes",
          "aN": 24,
          "bN": 12,
          "aContext": "Claude Code · instructions",
          "bContext": "Codex CLI · effort low · instructions"
        },
        {
          "metric": "What each call produced: strict pass, format miss, wrong values or error (Format miss)",
          "aValue": 17,
          "bValue": 0,
          "unit": "count",
          "aDisplay": "17",
          "bDisplay": "0",
          "winner": "unclear",
          "basis": "More or fewer count is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-outcomes",
          "aN": 24,
          "bN": 12,
          "aContext": "Claude Code · instructions",
          "bContext": "Codex CLI · effort low · instructions"
        },
        {
          "metric": "What each call produced: strict pass, format miss, wrong values or error (Wrong values)",
          "aValue": 7,
          "bValue": 0,
          "unit": "count",
          "aDisplay": "7",
          "bDisplay": "0",
          "winner": "unclear",
          "basis": "More or fewer count is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-outcomes",
          "aN": 24,
          "bN": 12,
          "aContext": "Claude Code · instructions",
          "bContext": "Codex CLI · effort low · instructions"
        },
        {
          "metric": "What each call produced: strict pass, format miss, wrong values or error (Error)",
          "aValue": 0,
          "bValue": 0,
          "unit": "count",
          "aDisplay": "0",
          "bDisplay": "0",
          "winner": "tie",
          "basis": "Same value. More or fewer is not better by itself for this metric.",
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-outcomes",
          "aN": 24,
          "bN": 12,
          "aContext": "Claude Code · instructions",
          "bContext": "Codex CLI · effort low · instructions"
        },
        {
          "metric": "Time per call, instructions vs schema mode",
          "aValue": 9.52,
          "bValue": 6.21,
          "unit": "seconds",
          "aDisplay": "9.52 s",
          "bDisplay": "6.21 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Haiku 4.5 5.67 s to 17.0 s; GPT-6.1 Sol (Codex CLI) 4.20 s to 12.3 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-time",
          "aN": 24,
          "bN": 12,
          "aContext": "Claude Code · instructions",
          "bContext": "Codex CLI · effort low · instructions",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            5.67,
            17
          ],
          "bRange": [
            4.2,
            12.27
          ]
        },
        {
          "metric": "Output and reasoning tokens per call, instructions vs schema mode (Median output tokens per call)",
          "aValue": 1128,
          "bValue": 117,
          "unit": "tokens",
          "aDisplay": "1,128",
          "bDisplay": "117",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "json-schema-vs-instructions",
          "chartId": "structured-output-tokens",
          "aN": 24,
          "bN": 12,
          "aContext": "Claude Code · instructions",
          "bContext": "Codex CLI · effort low · instructions"
        },
        {
          "metric": "Reasoning share of output tokens per call on hard tasks (calculation)",
          "aValue": 91.68,
          "bValue": 46.33,
          "unit": "percent",
          "aDisplay": "91.7%",
          "bDisplay": "46.3%",
          "winner": "unclear",
          "basis": "More or fewer percent is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-share",
          "aN": 24,
          "bN": 16,
          "aContext": "Claude Code",
          "bContext": "Codex CLI · effort medium",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            76.46,
            99.27
          ],
          "bRange": [
            11.42,
            86.85
          ],
          "calculation": true
        },
        {
          "metric": "List-price cost per call: reasoning, remaining output and input (calculation) (Reasoning (output tokens))",
          "aValue": 0.024492,
          "bValue": 0.002273,
          "unit": "usd",
          "aDisplay": "$0.024",
          "bDisplay": "$0.0023",
          "winner": "unclear",
          "basis": "More or fewer usd is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-cost-per-call",
          "aN": 24,
          "bN": 16,
          "aContext": "Claude Code",
          "bContext": "Codex CLI · effort medium",
          "calculation": true
        },
        {
          "metric": "List-price cost per call: reasoning, remaining output and input (calculation) (Remaining output (visible-answer estimate))",
          "aValue": 0.001795,
          "bValue": 0.003025,
          "unit": "usd",
          "aDisplay": "$0.0018",
          "bDisplay": "$0.0030",
          "winner": "unclear",
          "basis": "More or fewer usd is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-cost-per-call",
          "aN": 24,
          "bN": 16,
          "aContext": "Claude Code",
          "bContext": "Codex CLI · effort medium",
          "calculation": true
        },
        {
          "metric": "List-price cost per call: reasoning, remaining output and input (calculation) (Input (prompt, cache priced))",
          "aValue": 0.00451,
          "bValue": 0.020339,
          "unit": "usd",
          "aDisplay": "$0.0045",
          "bDisplay": "$0.020",
          "winner": "unclear",
          "basis": "More or fewer usd is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-cost-per-call",
          "aN": 24,
          "bN": 16,
          "aContext": "Claude Code",
          "bContext": "Codex CLI · effort medium",
          "calculation": true
        },
        {
          "metric": "Reasoning share on short tasks vs hard tasks (calculation) (Eight hard tasks)",
          "aValue": 91.68,
          "bValue": 46.33,
          "unit": "percent",
          "aDisplay": "91.7%",
          "bDisplay": "46.3%",
          "winner": "unclear",
          "basis": "More or fewer percent is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-short-vs-hard",
          "aN": 24,
          "bN": 16,
          "aContext": "Claude Code",
          "bContext": "Codex CLI · effort medium",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            76.46,
            99.27
          ],
          "bRange": [
            11.42,
            86.85
          ],
          "calculation": true
        },
        {
          "metric": "Reasoning share on short tasks vs hard tasks (calculation) (Five short tasks)",
          "aValue": 90.19,
          "bValue": 41.05,
          "unit": "percent",
          "aDisplay": "90.2%",
          "bDisplay": "41%",
          "winner": "unclear",
          "basis": "More or fewer percent is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "n": 15,
          "chartId": "thinking-bill-short-vs-hard",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code",
          "bContext": "Codex CLI · effort medium",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            73.1,
            97.59
          ],
          "bRange": [
            0,
            71.43
          ],
          "calculation": true
        },
        {
          "metric": "Time to first text: a 250-line answer, six models",
          "aValue": 4,
          "bValue": 3.52,
          "unit": "seconds",
          "aDisplay": "4.00 s",
          "bDisplay": "3.52 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Haiku 4.5 2.84 s to 6.38 s; GPT-6.1 Sol (Codex CLI) 2.75 s to 4.42 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "llm-speed-anatomy",
          "n": 4,
          "chartId": "speed-anatomy-first-text",
          "aN": 4,
          "bN": 4,
          "aContext": "Claude Code",
          "bContext": "Codex CLI · effort low",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            2.84,
            6.38
          ],
          "bRange": [
            2.75,
            4.42
          ]
        },
        {
          "metric": "Output speed after the first text: visible tokens per second (calculation)",
          "aValue": 153.2,
          "bValue": 79.6,
          "unit": "tokens",
          "aDisplay": "153",
          "bDisplay": "80",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "llm-speed-anatomy",
          "n": 4,
          "chartId": "speed-anatomy-output-speed",
          "aN": 4,
          "bN": 4,
          "aContext": "Claude Code",
          "bContext": "Codex CLI · effort low",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            152.6,
            216.1
          ],
          "bRange": [
            71.6,
            80.5
          ],
          "calculation": true
        },
        {
          "metric": "Output speed in characters per second after the first text (calculation)",
          "aValue": 547,
          "bValue": 323,
          "unit": "count",
          "aDisplay": "547",
          "bDisplay": "323",
          "winner": "unclear",
          "basis": "More or fewer count is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "llm-speed-anatomy",
          "chartId": "speed-anatomy-chars-per-second",
          "aN": 3,
          "bN": 4,
          "aContext": "Claude Code",
          "bContext": "Codex CLI · effort low",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            546,
            548
          ],
          "bRange": [
            291,
            327
          ],
          "calculation": true
        },
        {
          "metric": "Time to first text as the prompt grows: 1k",
          "aValue": 1.93,
          "bValue": 3.36,
          "unit": "seconds",
          "aDisplay": "1.93 s",
          "bDisplay": "3.36 s",
          "winner": "unclear",
          "basis": "Only 3 runs per side; the run ranges (fastest to slowest) do not overlap (Claude Haiku 4.5 1.85 s to 2.04 s; GPT-6.1 Sol (Codex CLI) 3.36 s to 4.75 s), but 3 runs cannot show a reliable difference. A range is not a confidence interval.",
          "studySlug": "llm-speed-anatomy",
          "n": 3,
          "chartId": "speed-anatomy-prompt-size",
          "aN": 3,
          "bN": 3,
          "aContext": "Claude Code",
          "bContext": "Codex CLI · effort low",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            1.85,
            2.04
          ],
          "bRange": [
            3.36,
            4.75
          ],
          "calculation": true
        },
        {
          "metric": "Time to first text as the prompt grows: 16k",
          "aValue": 2.27,
          "bValue": 4.02,
          "unit": "seconds",
          "aDisplay": "2.27 s",
          "bDisplay": "4.02 s",
          "winner": "unclear",
          "basis": "Only 3 runs per side; the run ranges (fastest to slowest) do not overlap (Claude Haiku 4.5 2.22 s to 2.47 s; GPT-6.1 Sol (Codex CLI) 3.30 s to 4.28 s), but 3 runs cannot show a reliable difference. A range is not a confidence interval.",
          "studySlug": "llm-speed-anatomy",
          "n": 3,
          "chartId": "speed-anatomy-prompt-size",
          "aN": 3,
          "bN": 3,
          "aContext": "Claude Code",
          "bContext": "Codex CLI · effort low",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            2.22,
            2.47
          ],
          "bRange": [
            3.3,
            4.28
          ],
          "calculation": true
        },
        {
          "metric": "Time to first text as the prompt grows: 64k",
          "aValue": 2.78,
          "bValue": 3.93,
          "unit": "seconds",
          "aDisplay": "2.78 s",
          "bDisplay": "3.93 s",
          "winner": "unclear",
          "basis": "Only 3 runs per side; the run ranges (fastest to slowest) do not overlap (Claude Haiku 4.5 2.45 s to 2.89 s; GPT-6.1 Sol (Codex CLI) 3.42 s to 4.38 s), but 3 runs cannot show a reliable difference. A range is not a confidence interval.",
          "studySlug": "llm-speed-anatomy",
          "n": 3,
          "chartId": "speed-anatomy-prompt-size",
          "aN": 3,
          "bN": 3,
          "aContext": "Claude Code",
          "bContext": "Codex CLI · effort low",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            2.45,
            2.89
          ],
          "bRange": [
            3.42,
            4.38
          ],
          "calculation": true
        },
        {
          "metric": "Total time per call by prompt size (1k prompt)",
          "aValue": 2.34,
          "bValue": 3.43,
          "unit": "seconds",
          "aDisplay": "2.34 s",
          "bDisplay": "3.43 s",
          "winner": "unclear",
          "basis": "Only 3 runs per side; the run ranges (fastest to slowest) do not overlap (Claude Haiku 4.5 2.22 s to 2.46 s; GPT-6.1 Sol (Codex CLI) 3.43 s to 4.92 s), but 3 runs cannot show a reliable difference. A range is not a confidence interval.",
          "studySlug": "llm-speed-anatomy",
          "n": 3,
          "chartId": "speed-anatomy-total-by-size",
          "aN": 3,
          "bN": 3,
          "aContext": "Claude Code",
          "bContext": "Codex CLI · effort low",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            2.22,
            2.46
          ],
          "bRange": [
            3.43,
            4.92
          ]
        },
        {
          "metric": "Total time per call by prompt size (16k prompt)",
          "aValue": 2.79,
          "bValue": 4.14,
          "unit": "seconds",
          "aDisplay": "2.79 s",
          "bDisplay": "4.14 s",
          "winner": "unclear",
          "basis": "Only 3 runs per side; the run ranges (fastest to slowest) do not overlap (Claude Haiku 4.5 2.58 s to 2.84 s; GPT-6.1 Sol (Codex CLI) 3.96 s to 4.68 s), but 3 runs cannot show a reliable difference. A range is not a confidence interval.",
          "studySlug": "llm-speed-anatomy",
          "n": 3,
          "chartId": "speed-anatomy-total-by-size",
          "aN": 3,
          "bN": 3,
          "aContext": "Claude Code",
          "bContext": "Codex CLI · effort low",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            2.58,
            2.84
          ],
          "bRange": [
            3.96,
            4.68
          ]
        },
        {
          "metric": "Total time per call by prompt size (64k prompt)",
          "aValue": 3.13,
          "bValue": 3.96,
          "unit": "seconds",
          "aDisplay": "3.13 s",
          "bDisplay": "3.96 s",
          "winner": "unclear",
          "basis": "Only 3 runs per side; the run ranges (fastest to slowest) do not overlap (Claude Haiku 4.5 2.84 s to 3.28 s; GPT-6.1 Sol (Codex CLI) 3.47 s to 4.44 s), but 3 runs cannot show a reliable difference. A range is not a confidence interval.",
          "studySlug": "llm-speed-anatomy",
          "n": 3,
          "chartId": "speed-anatomy-total-by-size",
          "aN": 3,
          "bN": 3,
          "aContext": "Claude Code",
          "bContext": "Codex CLI · effort low",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            2.84,
            3.28
          ],
          "bRange": [
            3.47,
            4.44
          ]
        },
        {
          "metric": "Exact lookup answers at the 1k, 16k and 64k prompt-size targets",
          "aValue": 1,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "100% (9/9)",
          "bDisplay": "100% (9/9)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Haiku 4.5 70% to 100%; GPT-6.1 Sol (Codex CLI) 70% to 100%), so this sample cannot separate them.",
          "studySlug": "llm-speed-anatomy",
          "n": 9,
          "chartId": "speed-anatomy-lookup-correct",
          "aN": 9,
          "bN": 9,
          "aContext": "Claude Code",
          "bContext": "Codex CLI · effort low",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.7009,
            1
          ],
          "bRange": [
            0.7009,
            1
          ]
        },
        {
          "metric": "Pass rate on 4 harder tasks (Strict pass)",
          "aValue": 0,
          "bValue": 0.6875,
          "unit": "rate",
          "aDisplay": "0% (0/12)",
          "bDisplay": "69% (11/16)",
          "winner": "b",
          "basis": "The 95% intervals do not overlap (Claude Haiku 4.5 0% to 24%; GPT-6.1 Sol (Codex CLI) 44% to 86%).",
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-pass-rate",
          "aN": 12,
          "bN": 16,
          "aContext": "Claude Code",
          "bContext": "Codex CLI · effort medium",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0,
            0.2425
          ],
          "bRange": [
            0.444,
            0.8584
          ]
        },
        {
          "metric": "Pass rate on 4 harder tasks (Lenient (format misses counted))",
          "aValue": 0,
          "bValue": 0.6875,
          "unit": "rate",
          "aDisplay": "0% (0/12)",
          "bDisplay": "69% (11/16)",
          "winner": "b",
          "basis": "The 95% intervals do not overlap (Claude Haiku 4.5 0% to 24%; GPT-6.1 Sol (Codex CLI) 44% to 86%).",
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-pass-rate",
          "aN": 12,
          "bN": 16,
          "aContext": "Claude Code",
          "bContext": "Codex CLI · effort medium",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0,
            0.2425
          ],
          "bRange": [
            0.444,
            0.8584
          ]
        },
        {
          "metric": "Calls that tried a tool although tools were off",
          "aValue": 0.0833,
          "bValue": 0,
          "unit": "rate",
          "aDisplay": "8% (1/12)",
          "bDisplay": "0% (0/16)",
          "winner": "unclear",
          "basis": "More or fewer rate is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-tool-attempts",
          "aN": 12,
          "bN": 16,
          "aContext": "Claude Code",
          "bContext": "Codex CLI · effort medium",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.0149,
            0.3539
          ],
          "bRange": [
            0,
            0.1936
          ]
        },
        {
          "metric": "Strict pass rate by task: 10x10 nonogram",
          "aValue": 0,
          "bValue": 0.75,
          "unit": "rate",
          "aDisplay": "0% (0/3)",
          "bDisplay": "75% (3/4)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Haiku 4.5 0% to 56%; GPT-6.1 Sol (Codex CLI) 30% to 95%), so this sample cannot separate them.",
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-pass-by-task",
          "aN": 3,
          "bN": 4,
          "aContext": "Claude Code",
          "bContext": "Codex CLI · effort medium",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0,
            0.5615
          ],
          "bRange": [
            0.3006,
            0.9544
          ]
        },
        {
          "metric": "Strict pass rate by task: Sudoku, 22 givens",
          "aValue": 0,
          "bValue": 0.25,
          "unit": "rate",
          "aDisplay": "0% (0/3)",
          "bDisplay": "25% (1/4)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Haiku 4.5 0% to 56%; GPT-6.1 Sol (Codex CLI) 5% to 70%), so this sample cannot separate them.",
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-pass-by-task",
          "aN": 3,
          "bN": 4,
          "aContext": "Claude Code",
          "bContext": "Codex CLI · effort medium",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0,
            0.5615
          ],
          "bRange": [
            0.0456,
            0.6994
          ]
        },
        {
          "metric": "Strict pass rate by task: 6x6 Skyscrapers",
          "aValue": 0,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "0% (0/3)",
          "bDisplay": "100% (4/4)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Haiku 4.5 0% to 56%; GPT-6.1 Sol (Codex CLI) 51% to 100%), so this sample cannot separate them.",
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-pass-by-task",
          "aN": 3,
          "bN": 4,
          "aContext": "Claude Code",
          "bContext": "Codex CLI · effort medium",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0,
            0.5615
          ],
          "bRange": [
            0.5101,
            1
          ]
        },
        {
          "metric": "Strict pass rate by task: Seeded shuffle output",
          "aValue": 0,
          "bValue": 0.75,
          "unit": "rate",
          "aDisplay": "0% (0/3)",
          "bDisplay": "75% (3/4)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Haiku 4.5 0% to 56%; GPT-6.1 Sol (Codex CLI) 30% to 95%), so this sample cannot separate them.",
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-pass-by-task",
          "aN": 3,
          "bN": 4,
          "aContext": "Claude Code",
          "bContext": "Codex CLI · effort medium",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0,
            0.5615
          ],
          "bRange": [
            0.3006,
            0.9544
          ]
        },
        {
          "metric": "Total time per call on harder tasks",
          "aValue": 108.98,
          "bValue": 120.24,
          "unit": "seconds",
          "aDisplay": "109.0 s",
          "bDisplay": "120.2 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Haiku 4.5 25.7 s to 223.9 s; GPT-6.1 Sol (Codex CLI) 46.2 s to 273.5 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-total-latency",
          "aN": 10,
          "bN": 13,
          "aContext": "Claude Code",
          "bContext": "Codex CLI · effort medium",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            25.73,
            223.95
          ],
          "bRange": [
            46.24,
            273.46
          ]
        },
        {
          "metric": "Output tokens per call on harder tasks (Output tokens)",
          "aValue": 12508,
          "bValue": 4994,
          "unit": "tokens",
          "aDisplay": "12,508",
          "bDisplay": "4,994",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "harder-tasks-head-to-head",
          "chartId": "harder-h2h-output-tokens",
          "aN": 10,
          "bN": 13,
          "aContext": "Claude Code",
          "bContext": "Codex CLI · effort medium",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            2965,
            26532
          ],
          "bRange": [
            2099,
            13413
          ]
        }
      ]
    },
    {
      "slug": "claude-fable-5-1-vs-gpt-6-1-sol-codex-cli",
      "a": "claude-fable-5-1",
      "b": "gpt-6-1-sol-codex-cli",
      "title": "Claude Fable 5.1 vs GPT-6.1 Sol (Codex CLI)",
      "seoTitle": "Claude Fable 5.1 vs GPT-6.1 Sol (Codex CLI): benchmarks",
      "description": "Claude Fable 5.1 vs GPT-6.1 Sol (Codex CLI): 12 measured metrics from 3 studies (Pass rate on five validated tasks; more), with sample sizes and intervals.",
      "verdict": "Claude Fable 5.1 and GPT-6.1 Sol (Codex CLI) share 12 measured metrics and 11 list-price calculations from 4 studies. No row separates them: every interval or run range overlaps, too few runs were recorded, no interval was recorded, or more is not better for that metric. The rows are 3 ties and 20 unclear; each row says why. Calculation rows are derived from list prices and recorded counts; they are not bills or runs. Every row ran the two sides through different routes (for example Claude Code vs Codex CLI), so they compare route + model pairs, not models alone; the contexts name the route. Some rows rest on small samples (n = 4 at the smallest).",
      "rows": [
        {
          "metric": "Pass rate on five validated tasks",
          "aValue": 1,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "100% (15/15)",
          "bDisplay": "100% (15/15)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Fable 5.1 80% to 100%; GPT-6.1 Sol (Codex CLI) 80% to 100%), so this sample cannot separate them.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-pass-rate",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code · five short validated tasks",
          "bContext": "Codex CLI · effort medium · five short validated tasks",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.7961,
            1
          ],
          "bRange": [
            0.7961,
            1
          ]
        },
        {
          "metric": "Total time per call",
          "aValue": 1.94,
          "bValue": 5.65,
          "unit": "seconds",
          "aDisplay": "1.94 s",
          "bDisplay": "5.65 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Fable 5.1 1.41 s to 9.83 s; GPT-6.1 Sol (Codex CLI) 4.10 s to 25.5 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-total-latency",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code · five short validated tasks",
          "bContext": "Codex CLI · effort medium · five short validated tasks",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            1.41,
            9.83
          ],
          "bRange": [
            4.1,
            25.46
          ]
        },
        {
          "metric": "Time to first useful output",
          "aValue": 1.2,
          "bValue": 5.05,
          "unit": "seconds",
          "aDisplay": "1.20 s",
          "bDisplay": "5.05 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Fable 5.1 0.95 s to 7.90 s; GPT-6.1 Sol (Codex CLI) 3.36 s to 17.8 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-first-useful-latency",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code · five short validated tasks",
          "bContext": "Codex CLI · effort medium · five short validated tasks",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            0.95,
            7.9
          ],
          "bRange": [
            3.36,
            17.82
          ]
        },
        {
          "metric": "Input tokens per call: what the CLI sends (Cache read)",
          "aValue": 2760,
          "bValue": 5180,
          "unit": "tokens",
          "aDisplay": "2,760",
          "bDisplay": "5,180",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-input-tokens",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code · five short validated tasks",
          "bContext": "Codex CLI · effort medium · five short validated tasks"
        },
        {
          "metric": "Input tokens per call: what the CLI sends (Other input)",
          "aValue": 473,
          "bValue": 6943,
          "unit": "tokens",
          "aDisplay": "473",
          "bDisplay": "6,943",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-input-tokens",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code · five short validated tasks",
          "bContext": "Codex CLI · effort medium · five short validated tasks"
        },
        {
          "metric": "Output tokens per call (Output tokens)",
          "aValue": 64,
          "bValue": 42,
          "unit": "tokens",
          "aDisplay": "64",
          "bDisplay": "42",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-output-tokens",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code · five short validated tasks",
          "bContext": "Codex CLI · effort medium · five short validated tasks"
        },
        {
          "metric": "List-price cost per call (calculation)",
          "aValue": 0.00987,
          "bValue": 0.01018,
          "unit": "usd",
          "aDisplay": "$0.0099",
          "bDisplay": "$0.010",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Fable 5.1 $0.0049 to $0.058; GPT-6.1 Sol (Codex CLI) $0.0054 to $0.027); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-list-price-per-call",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code · five short validated tasks",
          "bContext": "Codex CLI · effort medium · five short validated tasks",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            0.0049,
            0.05843
          ],
          "bRange": [
            0.0054,
            0.02686
          ],
          "calculation": true
        },
        {
          "metric": "List-price cost per passing answer (calculation)",
          "aValue": 0.02054,
          "bValue": 0.01564,
          "unit": "usd",
          "aDisplay": "$0.021",
          "bDisplay": "$0.016",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.021 vs $0.016) is not tested against run-to-run variation.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-cost-per-pass",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code · five short validated tasks",
          "bContext": "Codex CLI · effort medium · five short validated tasks",
          "calculation": true
        },
        {
          "metric": "Pass rate on eight hard tasks (Strict pass)",
          "aValue": 1,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "100% (24/24)",
          "bDisplay": "100% (16/16)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Fable 5.1 86% to 100%; GPT-6.1 Sol (Codex CLI) 81% to 100%), so this sample cannot separate them.",
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-pass-rate",
          "aN": 24,
          "bN": 16,
          "aContext": "Claude Code · eight hard validated tasks",
          "bContext": "Codex CLI · effort medium · eight hard validated tasks",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.862,
            1
          ],
          "bRange": [
            0.8064,
            1
          ]
        },
        {
          "metric": "Pass rate on eight hard tasks (Lenient (format misses counted))",
          "aValue": 1,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "100% (24/24)",
          "bDisplay": "100% (16/16)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Fable 5.1 86% to 100%; GPT-6.1 Sol (Codex CLI) 81% to 100%), so this sample cannot separate them.",
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-pass-rate",
          "aN": 24,
          "bN": 16,
          "aContext": "Claude Code · eight hard validated tasks",
          "bContext": "Codex CLI · effort medium · eight hard validated tasks",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.862,
            1
          ],
          "bRange": [
            0.8064,
            1
          ]
        },
        {
          "metric": "Total time per call on hard tasks (separate batches)",
          "aValue": 16.13,
          "bValue": 13.11,
          "unit": "seconds",
          "aDisplay": "16.1 s",
          "bDisplay": "13.1 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Fable 5.1 4.46 s to 90.0 s; GPT-6.1 Sol (Codex CLI) 8.54 s to 61.6 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-total-latency",
          "aN": 24,
          "bN": 16,
          "aContext": "Claude Code · eight hard validated tasks",
          "bContext": "Codex CLI · effort medium · eight hard validated tasks",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            4.46,
            90
          ],
          "bRange": [
            8.54,
            61.6
          ]
        },
        {
          "metric": "Time to first useful output on hard tasks",
          "aValue": 11.63,
          "bValue": 10.23,
          "unit": "seconds",
          "aDisplay": "11.6 s",
          "bDisplay": "10.2 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Fable 5.1 2.00 s to 85.3 s; GPT-6.1 Sol (Codex CLI) 6.09 s to 40.4 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-first-useful-latency",
          "aN": 24,
          "bN": 16,
          "aContext": "Claude Code · eight hard validated tasks",
          "bContext": "Codex CLI · effort medium · eight hard validated tasks",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            2,
            85.33
          ],
          "bRange": [
            6.09,
            40.41
          ]
        },
        {
          "metric": "Output tokens per call on hard tasks (Output tokens)",
          "aValue": 1366,
          "bValue": 335,
          "unit": "tokens",
          "aDisplay": "1,366",
          "bDisplay": "335",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-output-tokens",
          "aN": 24,
          "bN": 16,
          "aContext": "Claude Code · eight hard validated tasks",
          "bContext": "Codex CLI · effort medium · eight hard validated tasks"
        },
        {
          "metric": "List-price cost per strict pass on hard tasks (calculation)",
          "aValue": 0.09331,
          "bValue": 0.02564,
          "unit": "usd",
          "aDisplay": "$0.093",
          "bDisplay": "$0.026",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.093 vs $0.026, 3.6x) is not tested against run-to-run variation.",
          "studySlug": "hard-model-head-to-head",
          "chartId": "hard-h2h-cost-per-pass",
          "aN": 24,
          "bN": 16,
          "aContext": "Claude Code · eight hard validated tasks",
          "bContext": "Codex CLI · effort medium · eight hard validated tasks",
          "calculation": true
        },
        {
          "metric": "Reasoning share of output tokens per call on hard tasks (calculation)",
          "aValue": 64.24,
          "bValue": 46.33,
          "unit": "percent",
          "aDisplay": "64.2%",
          "bDisplay": "46.3%",
          "winner": "unclear",
          "basis": "More or fewer percent is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-share",
          "aN": 24,
          "bN": 16,
          "aContext": "Claude Code",
          "bContext": "Codex CLI · effort medium",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            23.44,
            97.19
          ],
          "bRange": [
            11.42,
            86.85
          ],
          "calculation": true
        },
        {
          "metric": "List-price cost per call: reasoning, remaining output and input (calculation) (Reasoning (output tokens))",
          "aValue": 0.053696,
          "bValue": 0.002273,
          "unit": "usd",
          "aDisplay": "$0.054",
          "bDisplay": "$0.0023",
          "winner": "unclear",
          "basis": "More or fewer usd is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-cost-per-call",
          "aN": 24,
          "bN": 16,
          "aContext": "Claude Code",
          "bContext": "Codex CLI · effort medium",
          "calculation": true
        },
        {
          "metric": "List-price cost per call: reasoning, remaining output and input (calculation) (Remaining output (visible-answer estimate))",
          "aValue": 0.018702,
          "bValue": 0.003025,
          "unit": "usd",
          "aDisplay": "$0.019",
          "bDisplay": "$0.0030",
          "winner": "unclear",
          "basis": "More or fewer usd is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-cost-per-call",
          "aN": 24,
          "bN": 16,
          "aContext": "Claude Code",
          "bContext": "Codex CLI · effort medium",
          "calculation": true
        },
        {
          "metric": "List-price cost per call: reasoning, remaining output and input (calculation) (Input (prompt, cache priced))",
          "aValue": 0.02091,
          "bValue": 0.020339,
          "unit": "usd",
          "aDisplay": "$0.021",
          "bDisplay": "$0.020",
          "winner": "unclear",
          "basis": "More or fewer usd is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-cost-per-call",
          "aN": 24,
          "bN": 16,
          "aContext": "Claude Code",
          "bContext": "Codex CLI · effort medium",
          "calculation": true
        },
        {
          "metric": "Reasoning share on short tasks vs hard tasks (calculation) (Eight hard tasks)",
          "aValue": 64.24,
          "bValue": 46.33,
          "unit": "percent",
          "aDisplay": "64.2%",
          "bDisplay": "46.3%",
          "winner": "unclear",
          "basis": "More or fewer percent is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "chartId": "thinking-bill-short-vs-hard",
          "aN": 24,
          "bN": 16,
          "aContext": "Claude Code",
          "bContext": "Codex CLI · effort medium",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            23.44,
            97.19
          ],
          "bRange": [
            11.42,
            86.85
          ],
          "calculation": true
        },
        {
          "metric": "Reasoning share on short tasks vs hard tasks (calculation) (Five short tasks)",
          "aValue": 0,
          "bValue": 41.05,
          "unit": "percent",
          "aDisplay": "0%",
          "bDisplay": "41%",
          "winner": "unclear",
          "basis": "More or fewer percent is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "n": 15,
          "chartId": "thinking-bill-short-vs-hard",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code",
          "bContext": "Codex CLI · effort medium",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            0,
            74.01
          ],
          "bRange": [
            0,
            71.43
          ],
          "calculation": true
        },
        {
          "metric": "Time to first text: a 250-line answer, six models",
          "aValue": 4.43,
          "bValue": 3.52,
          "unit": "seconds",
          "aDisplay": "4.43 s",
          "bDisplay": "3.52 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Fable 5.1 2.27 s to 4.64 s; GPT-6.1 Sol (Codex CLI) 2.75 s to 4.42 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "llm-speed-anatomy",
          "n": 4,
          "chartId": "speed-anatomy-first-text",
          "aN": 4,
          "bN": 4,
          "aContext": "Claude Code",
          "bContext": "Codex CLI · effort low",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            2.27,
            4.64
          ],
          "bRange": [
            2.75,
            4.42
          ]
        },
        {
          "metric": "Output speed after the first text: visible tokens per second (calculation)",
          "aValue": 122.6,
          "bValue": 79.6,
          "unit": "tokens",
          "aDisplay": "123",
          "bDisplay": "80",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "llm-speed-anatomy",
          "n": 4,
          "chartId": "speed-anatomy-output-speed",
          "aN": 4,
          "bN": 4,
          "aContext": "Claude Code",
          "bContext": "Codex CLI · effort low",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            120.9,
            131.4
          ],
          "bRange": [
            71.6,
            80.5
          ],
          "calculation": true
        },
        {
          "metric": "Output speed in characters per second after the first text (calculation)",
          "aValue": 273,
          "bValue": 323,
          "unit": "count",
          "aDisplay": "273",
          "bDisplay": "323",
          "winner": "unclear",
          "basis": "More or fewer count is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "llm-speed-anatomy",
          "n": 4,
          "chartId": "speed-anatomy-chars-per-second",
          "aN": 4,
          "bN": 4,
          "aContext": "Claude Code",
          "bContext": "Codex CLI · effort low",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            270,
            293
          ],
          "bRange": [
            291,
            327
          ],
          "calculation": true
        }
      ]
    },
    {
      "slug": "jev-1-13-vs-deterministic-routing-policy",
      "a": "jev-1-13",
      "b": "deterministic-routing-policy",
      "title": "Jev 1.13 vs Deterministic routing policy",
      "seoTitle": "Jev 1.13 vs Deterministic routing policy: benchmarks",
      "description": "Jev 1.13 vs Deterministic routing policy: 3 measured metrics from one study (Time to make one routing decision; more), with sample sizes and intervals.",
      "verdict": "Jev 1.13 and Deterministic routing policy share 3 measured metrics and 4 list-price calculations from one study. Deterministic routing policy leads on 1 row: Time to make one routing decision, 1.42 µs vs 137 ms. On those rows the p50–p95 bands do not overlap; only a 95% interval is a confidence interval. The other rows are 1 tie and 5 unclear; each row says why. Calculation rows are derived from list prices and recorded counts; they are not bills or runs.",
      "rows": [
        {
          "metric": "Time to make one routing decision",
          "aValue": 136.5,
          "bValue": 0.00142,
          "unit": "ms",
          "aDisplay": "137 ms",
          "bDisplay": "1.42 µs",
          "winner": "b",
          "basis": "Jev 1.13’s median is above Deterministic routing policy’s 95th percentile (p50–p95 bands: Jev 1.13 137 ms to 196 ms; Deterministic routing policy 1.42 µs to 2.33 µs); not a confidence interval.",
          "studySlug": "routing-overhead",
          "chartId": "router-overhead-decision-latency",
          "aN": 246,
          "bN": 20000,
          "aContext": "routing overhead per decision · TypeSafe API",
          "bContext": "Agent · in process · routing overhead per decision",
          "rangeKind": "range",
          "spanKind": "p50-p95",
          "aRange": [
            136.5,
            195.7
          ],
          "bRange": [
            0.00142,
            0.00233
          ]
        },
        {
          "metric": "Routing calls that returned a decision",
          "aValue": 1,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "100% (246/246)",
          "bDisplay": "100% (20000/20000)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Jev 1.13 98% to 100%; Deterministic routing policy 100% to 100%), so this sample cannot separate them.",
          "studySlug": "routing-overhead",
          "chartId": "router-overhead-completed",
          "aN": 246,
          "bN": 20000,
          "aContext": "routing overhead per decision · TypeSafe API",
          "bContext": "Agent · in process · routing overhead per decision",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.9846,
            1
          ],
          "bRange": [
            0.9998,
            1
          ]
        },
        {
          "metric": "Cost per 1,000 routing decisions: no model call vs provider-reported",
          "aValue": 0.0337,
          "bValue": 0,
          "unit": "usd",
          "aDisplay": "$0.034",
          "bDisplay": "$0.00",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.034 vs $0.00) is not tested against run-to-run variation.",
          "studySlug": "routing-overhead",
          "chartId": "router-overhead-cost-reported",
          "aN": 82,
          "bN": 20000,
          "aContext": "routing overhead per decision · TypeSafe API",
          "bContext": "Agent · in process · routing overhead per decision"
        },
        {
          "metric": "Added routing cost per 1,000 tasks (calculation) (Every model call routed (49.5 per task))",
          "aValue": 1.67,
          "bValue": 0,
          "unit": "usd",
          "aDisplay": "$1.67",
          "bDisplay": "$0.00",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($1.67 vs $0.00) is not tested against run-to-run variation.",
          "studySlug": "routing-overhead",
          "chartId": "router-overhead-cost-per-1000-tasks",
          "aContext": "calculation per 1,000 tasks from recorded decision counts · TypeSafe API",
          "bContext": "Agent · in process · calculation per 1,000 tasks from recorded decision counts",
          "calculation": true
        },
        {
          "metric": "Added routing cost per 1,000 tasks (calculation) (Only System One decisions (7 per task))",
          "aValue": 0.24,
          "bValue": 0,
          "unit": "usd",
          "aDisplay": "$0.24",
          "bDisplay": "$0.00",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.24 vs $0.00) is not tested against run-to-run variation.",
          "studySlug": "routing-overhead",
          "chartId": "router-overhead-cost-per-1000-tasks",
          "aContext": "calculation per 1,000 tasks from recorded decision counts · TypeSafe API",
          "bContext": "Agent · in process · calculation per 1,000 tasks from recorded decision counts",
          "calculation": true
        },
        {
          "metric": "Added routing delay per task (calculation) (Every model call routed (49.5 per task))",
          "aValue": 6.7568,
          "bValue": 0.0000703,
          "unit": "seconds",
          "aDisplay": "6.76 s",
          "bDisplay": "70.3 µs",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap (6.76 s vs 70.3 µs, 96,114x) is not tested against run-to-run variation.",
          "studySlug": "routing-overhead",
          "chartId": "router-overhead-delay-per-task",
          "aContext": "calculation per task from recorded decision counts, decisions in line · TypeSafe API",
          "bContext": "Agent · in process · calculation per task from recorded decision counts, decisions in line",
          "calculation": true
        },
        {
          "metric": "Added routing delay per task (calculation) (Only System One decisions (7 per task))",
          "aValue": 0.9555,
          "bValue": 0.0000099,
          "unit": "seconds",
          "aDisplay": "0.96 s",
          "bDisplay": "9.9 µs",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap (0.96 s vs 9.9 µs, 96,515x) is not tested against run-to-run variation.",
          "studySlug": "routing-overhead",
          "chartId": "router-overhead-delay-per-task",
          "aContext": "calculation per task from recorded decision counts, decisions in line · TypeSafe API",
          "bContext": "Agent · in process · calculation per task from recorded decision counts, decisions in line",
          "calculation": true
        }
      ]
    },
    {
      "slug": "deterministic-routing-policy-vs-claude-sonnet-5-5",
      "a": "deterministic-routing-policy",
      "b": "claude-sonnet-5-5",
      "title": "Deterministic routing policy vs Claude Sonnet 5.5",
      "seoTitle": "Deterministic routing policy vs Claude Sonnet 5.5",
      "description": "Deterministic routing policy vs Claude Sonnet 5.5: 2 measured metrics from one study (Time to make one routing decision; more), with sample sizes and intervals.",
      "verdict": "Deterministic routing policy and Claude Sonnet 5.5 share 2 measured metrics and 4 list-price calculations from one study. Deterministic routing policy leads on 1 row: Time to make one routing decision, 1.42 µs vs 2,597 ms. On those rows the p50–p95 bands do not overlap; only a 95% interval is a confidence interval. The other rows are 1 tie and 4 unclear; each row says why. Calculation rows are derived from list prices and recorded counts; they are not bills or runs.",
      "rows": [
        {
          "metric": "Time to make one routing decision",
          "aValue": 0.00142,
          "bValue": 2597,
          "unit": "ms",
          "aDisplay": "1.42 µs",
          "bDisplay": "2,597 ms",
          "winner": "a",
          "basis": "Claude Sonnet 5.5’s median is above Deterministic routing policy’s 95th percentile (p50–p95 bands: Deterministic routing policy 1.42 µs to 2.33 µs; Claude Sonnet 5.5 2,597 ms to 4,298 ms); not a confidence interval.",
          "studySlug": "routing-overhead",
          "chartId": "router-overhead-decision-latency",
          "aN": 20000,
          "bN": 82,
          "aContext": "Agent · in process · routing overhead per decision",
          "bContext": "effort low · via Claude Code · routing overhead per decision",
          "rangeKind": "range",
          "spanKind": "p50-p95",
          "aRange": [
            0.00142,
            0.00233
          ],
          "bRange": [
            2597,
            4298
          ]
        },
        {
          "metric": "Routing calls that returned a decision",
          "aValue": 1,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "100% (20000/20000)",
          "bDisplay": "100% (82/82)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Deterministic routing policy 100% to 100%; Claude Sonnet 5.5 96% to 100%), so this sample cannot separate them.",
          "studySlug": "routing-overhead",
          "chartId": "router-overhead-completed",
          "aN": 20000,
          "bN": 82,
          "aContext": "Agent · in process · routing overhead per decision",
          "bContext": "effort low · via Claude Code · routing overhead per decision",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.9998,
            1
          ],
          "bRange": [
            0.9552,
            1
          ]
        },
        {
          "metric": "Added routing cost per 1,000 tasks (calculation) (Every model call routed (49.5 per task))",
          "aValue": 0,
          "bValue": 247.3,
          "unit": "usd",
          "aDisplay": "$0.00",
          "bDisplay": "$247.30",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.00 vs $247.30) is not tested against run-to-run variation.",
          "studySlug": "routing-overhead",
          "chartId": "router-overhead-cost-per-1000-tasks",
          "aContext": "Agent · in process · calculation per 1,000 tasks from recorded decision counts",
          "bContext": "effort low · via Claude Code · calculation per 1,000 tasks from recorded decision counts",
          "calculation": true
        },
        {
          "metric": "Added routing cost per 1,000 tasks (calculation) (Only System One decisions (7 per task))",
          "aValue": 0,
          "bValue": 34.97,
          "unit": "usd",
          "aDisplay": "$0.00",
          "bDisplay": "$34.97",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.00 vs $34.97) is not tested against run-to-run variation.",
          "studySlug": "routing-overhead",
          "chartId": "router-overhead-cost-per-1000-tasks",
          "aContext": "Agent · in process · calculation per 1,000 tasks from recorded decision counts",
          "bContext": "effort low · via Claude Code · calculation per 1,000 tasks from recorded decision counts",
          "calculation": true
        },
        {
          "metric": "Added routing delay per task (calculation) (Every model call routed (49.5 per task))",
          "aValue": 0.0000703,
          "bValue": 128.5515,
          "unit": "seconds",
          "aDisplay": "70.3 µs",
          "bDisplay": "128.6 s",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap (70.3 µs vs 128.6 s, 1.8 million times) is not tested against run-to-run variation.",
          "studySlug": "routing-overhead",
          "chartId": "router-overhead-delay-per-task",
          "aContext": "Agent · in process · calculation per task from recorded decision counts, decisions in line",
          "bContext": "effort low · via Claude Code · calculation per task from recorded decision counts, decisions in line",
          "calculation": true
        },
        {
          "metric": "Added routing delay per task (calculation) (Only System One decisions (7 per task))",
          "aValue": 0.0000099,
          "bValue": 18.179,
          "unit": "seconds",
          "aDisplay": "9.9 µs",
          "bDisplay": "18.2 s",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap (9.9 µs vs 18.2 s, 1.8 million times) is not tested against run-to-run variation.",
          "studySlug": "routing-overhead",
          "chartId": "router-overhead-delay-per-task",
          "aContext": "Agent · in process · calculation per task from recorded decision counts, decisions in line",
          "bContext": "effort low · via Claude Code · calculation per task from recorded decision counts, decisions in line",
          "calculation": true
        }
      ]
    },
    {
      "slug": "deterministic-routing-policy-vs-claude-haiku-4-5",
      "a": "deterministic-routing-policy",
      "b": "claude-haiku-4-5",
      "title": "Deterministic routing policy vs Claude Haiku 4.5",
      "seoTitle": "Deterministic routing policy vs Claude Haiku 4.5: benchmarks",
      "description": "Deterministic routing policy vs Claude Haiku 4.5: 2 measured metrics from one study (Time to make one routing decision; more), with sample sizes and intervals.",
      "verdict": "Deterministic routing policy and Claude Haiku 4.5 share 2 measured metrics and 4 list-price calculations from one study. Deterministic routing policy leads on 1 row: Time to make one routing decision, 1.42 µs vs 12,543 ms. On those rows the p50–p95 bands do not overlap; only a 95% interval is a confidence interval. The other rows are 1 tie and 4 unclear; each row says why. Calculation rows are derived from list prices and recorded counts; they are not bills or runs.",
      "rows": [
        {
          "metric": "Time to make one routing decision",
          "aValue": 0.00142,
          "bValue": 12543,
          "unit": "ms",
          "aDisplay": "1.42 µs",
          "bDisplay": "12,543 ms",
          "winner": "a",
          "basis": "Claude Haiku 4.5’s median is above Deterministic routing policy’s 95th percentile (p50–p95 bands: Deterministic routing policy 1.42 µs to 2.33 µs; Claude Haiku 4.5 12,543 ms to 34,481 ms); not a confidence interval.",
          "studySlug": "routing-overhead",
          "chartId": "router-overhead-decision-latency",
          "aN": 20000,
          "bN": 82,
          "aContext": "Agent · in process · routing overhead per decision",
          "bContext": "thinking on · via Claude Code · routing overhead per decision",
          "rangeKind": "range",
          "spanKind": "p50-p95",
          "aRange": [
            0.00142,
            0.00233
          ],
          "bRange": [
            12543,
            34481
          ]
        },
        {
          "metric": "Routing calls that returned a decision",
          "aValue": 1,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "100% (20000/20000)",
          "bDisplay": "100% (82/82)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Deterministic routing policy 100% to 100%; Claude Haiku 4.5 96% to 100%), so this sample cannot separate them.",
          "studySlug": "routing-overhead",
          "chartId": "router-overhead-completed",
          "aN": 20000,
          "bN": 82,
          "aContext": "Agent · in process · routing overhead per decision",
          "bContext": "thinking on · via Claude Code · routing overhead per decision",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.9998,
            1
          ],
          "bRange": [
            0.9552,
            1
          ]
        },
        {
          "metric": "Added routing cost per 1,000 tasks (calculation) (Every model call routed (49.5 per task))",
          "aValue": 0,
          "bValue": 441.74,
          "unit": "usd",
          "aDisplay": "$0.00",
          "bDisplay": "$441.74",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.00 vs $441.74) is not tested against run-to-run variation.",
          "studySlug": "routing-overhead",
          "chartId": "router-overhead-cost-per-1000-tasks",
          "aContext": "Agent · in process · calculation per 1,000 tasks from recorded decision counts",
          "bContext": "thinking on · via Claude Code · calculation per 1,000 tasks from recorded decision counts",
          "calculation": true
        },
        {
          "metric": "Added routing cost per 1,000 tasks (calculation) (Only System One decisions (7 per task))",
          "aValue": 0,
          "bValue": 62.47,
          "unit": "usd",
          "aDisplay": "$0.00",
          "bDisplay": "$62.47",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.00 vs $62.47) is not tested against run-to-run variation.",
          "studySlug": "routing-overhead",
          "chartId": "router-overhead-cost-per-1000-tasks",
          "aContext": "Agent · in process · calculation per 1,000 tasks from recorded decision counts",
          "bContext": "thinking on · via Claude Code · calculation per 1,000 tasks from recorded decision counts",
          "calculation": true
        },
        {
          "metric": "Added routing delay per task (calculation) (Every model call routed (49.5 per task))",
          "aValue": 0.0000703,
          "bValue": 620.8785,
          "unit": "seconds",
          "aDisplay": "70.3 µs",
          "bDisplay": "620.9 s",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap (70.3 µs vs 620.9 s, 8.8 million times) is not tested against run-to-run variation.",
          "studySlug": "routing-overhead",
          "chartId": "router-overhead-delay-per-task",
          "aContext": "Agent · in process · calculation per task from recorded decision counts, decisions in line",
          "bContext": "thinking on · via Claude Code · calculation per task from recorded decision counts, decisions in line",
          "calculation": true
        },
        {
          "metric": "Added routing delay per task (calculation) (Only System One decisions (7 per task))",
          "aValue": 0.0000099,
          "bValue": 87.801,
          "unit": "seconds",
          "aDisplay": "9.9 µs",
          "bDisplay": "87.8 s",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap (9.9 µs vs 87.8 s, 8.9 million times) is not tested against run-to-run variation.",
          "studySlug": "routing-overhead",
          "chartId": "router-overhead-delay-per-task",
          "aContext": "Agent · in process · calculation per task from recorded decision counts, decisions in line",
          "bContext": "thinking on · via Claude Code · calculation per task from recorded decision counts, decisions in line",
          "calculation": true
        }
      ]
    },
    {
      "slug": "openrouter-vs-anthropic",
      "a": "openrouter",
      "b": "anthropic",
      "title": "OpenRouter vs Anthropic",
      "seoTitle": "OpenRouter vs Anthropic: price per model",
      "description": "OpenRouter vs Anthropic: 14 list prices per million tokens for the same models, as reported by OpenRouter’s public API, with the gap stated.",
      "verdict": "OpenRouter and Anthropic share 14 reported list prices from one study. No row names a winner: a reported list price has no interval, so each gap is stated, not ranked. Prices change often; the snapshot date is in each row’s context. The rows are 14 ties; each row says why.",
      "rows": [
        {
          "metric": "Claude Haiku 4.5: OpenRouter vs Anthropic list price (Input)",
          "aValue": 1,
          "bValue": 1,
          "unit": "usd",
          "aDisplay": "$1.00",
          "bDisplay": "$1.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "gateway-vs-direct-claude-haiku-4-5",
          "aContext": "Claude Haiku 4.5 · list price, snapshot 2026-10-06",
          "bContext": "first-party list price · Claude Haiku 4.5 · list price, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Haiku 4.5: OpenRouter vs Anthropic list price (Output)",
          "aValue": 5,
          "bValue": 5,
          "unit": "usd",
          "aDisplay": "$5.00",
          "bDisplay": "$5.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "gateway-vs-direct-claude-haiku-4-5",
          "aContext": "Claude Haiku 4.5 · list price, snapshot 2026-10-06",
          "bContext": "first-party list price · Claude Haiku 4.5 · list price, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Sonnet 5: OpenRouter vs Anthropic list price (Input)",
          "aValue": 2,
          "bValue": 2,
          "unit": "usd",
          "aDisplay": "$2.00",
          "bDisplay": "$2.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "gateway-vs-direct-claude-sonnet-5",
          "aContext": "Claude Sonnet 5 · list price, snapshot 2026-10-06",
          "bContext": "first-party list price · Claude Sonnet 5 · list price, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Sonnet 5: OpenRouter vs Anthropic list price (Output)",
          "aValue": 10,
          "bValue": 10,
          "unit": "usd",
          "aDisplay": "$10.00",
          "bDisplay": "$10.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "gateway-vs-direct-claude-sonnet-5",
          "aContext": "Claude Sonnet 5 · list price, snapshot 2026-10-06",
          "bContext": "first-party list price · Claude Sonnet 5 · list price, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Sonnet 5.5: OpenRouter vs Anthropic list price (Input)",
          "aValue": 2,
          "bValue": 2,
          "unit": "usd",
          "aDisplay": "$2.00",
          "bDisplay": "$2.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "gateway-vs-direct-claude-sonnet-5-5",
          "aContext": "Claude Sonnet 5.5 · list price, snapshot 2026-10-06",
          "bContext": "first-party list price · Claude Sonnet 5.5 · list price, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Sonnet 5.5: OpenRouter vs Anthropic list price (Output)",
          "aValue": 10,
          "bValue": 10,
          "unit": "usd",
          "aDisplay": "$10.00",
          "bDisplay": "$10.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "gateway-vs-direct-claude-sonnet-5-5",
          "aContext": "Claude Sonnet 5.5 · list price, snapshot 2026-10-06",
          "bContext": "first-party list price · Claude Sonnet 5.5 · list price, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Opus 4.8: OpenRouter vs Anthropic list price (Input)",
          "aValue": 5,
          "bValue": 5,
          "unit": "usd",
          "aDisplay": "$5.00",
          "bDisplay": "$5.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "gateway-vs-direct-claude-opus-4-8",
          "aContext": "Claude Opus 4.8 · list price, snapshot 2026-10-06",
          "bContext": "first-party list price · Claude Opus 4.8 · list price, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Opus 4.8: OpenRouter vs Anthropic list price (Output)",
          "aValue": 25,
          "bValue": 25,
          "unit": "usd",
          "aDisplay": "$25.00",
          "bDisplay": "$25.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "gateway-vs-direct-claude-opus-4-8",
          "aContext": "Claude Opus 4.8 · list price, snapshot 2026-10-06",
          "bContext": "first-party list price · Claude Opus 4.8 · list price, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Opus 5: OpenRouter vs Anthropic list price (Input)",
          "aValue": 5,
          "bValue": 5,
          "unit": "usd",
          "aDisplay": "$5.00",
          "bDisplay": "$5.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "gateway-vs-direct-claude-opus-5",
          "aContext": "Claude Opus 5 · list price, snapshot 2026-10-06",
          "bContext": "first-party list price · Claude Opus 5 · list price, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Opus 5: OpenRouter vs Anthropic list price (Output)",
          "aValue": 25,
          "bValue": 25,
          "unit": "usd",
          "aDisplay": "$25.00",
          "bDisplay": "$25.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "gateway-vs-direct-claude-opus-5",
          "aContext": "Claude Opus 5 · list price, snapshot 2026-10-06",
          "bContext": "first-party list price · Claude Opus 5 · list price, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Opus 5.5: OpenRouter vs Anthropic list price (Input)",
          "aValue": 4,
          "bValue": 4,
          "unit": "usd",
          "aDisplay": "$4.00",
          "bDisplay": "$4.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "gateway-vs-direct-claude-opus-5-5",
          "aContext": "Claude Opus 5.5 · list price, snapshot 2026-10-06",
          "bContext": "first-party list price · Claude Opus 5.5 · list price, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Opus 5.5: OpenRouter vs Anthropic list price (Output)",
          "aValue": 20,
          "bValue": 20,
          "unit": "usd",
          "aDisplay": "$20.00",
          "bDisplay": "$20.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "gateway-vs-direct-claude-opus-5-5",
          "aContext": "Claude Opus 5.5 · list price, snapshot 2026-10-06",
          "bContext": "first-party list price · Claude Opus 5.5 · list price, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Fable 5.1: OpenRouter vs Anthropic list price (Input)",
          "aValue": 10,
          "bValue": 10,
          "unit": "usd",
          "aDisplay": "$10.00",
          "bDisplay": "$10.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "gateway-vs-direct-claude-fable-5-1",
          "aContext": "Claude Fable 5.1 · list price, snapshot 2026-10-06",
          "bContext": "first-party list price · Claude Fable 5.1 · list price, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Fable 5.1: OpenRouter vs Anthropic list price (Output)",
          "aValue": 50,
          "bValue": 50,
          "unit": "usd",
          "aDisplay": "$50.00",
          "bDisplay": "$50.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "gateway-vs-direct-claude-fable-5-1",
          "aContext": "Claude Fable 5.1 · list price, snapshot 2026-10-06",
          "bContext": "first-party list price · Claude Fable 5.1 · list price, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "openrouter-vs-google-ai-studio",
      "a": "openrouter",
      "b": "google-ai-studio",
      "title": "OpenRouter vs Google AI Studio",
      "seoTitle": "OpenRouter vs Google AI Studio: price per model",
      "description": "OpenRouter vs Google AI Studio: 4 list prices per million tokens for the same models, as reported by OpenRouter’s public API, with the gap stated.",
      "verdict": "OpenRouter and Google AI Studio share 4 reported list prices from one study. No row names a winner: a reported list price has no interval, so each gap is stated, not ranked. Prices change often; the snapshot date is in each row’s context. The rows are 4 ties; each row says why.",
      "rows": [
        {
          "metric": "Gemini 3.8 Flash: OpenRouter vs Google list price (Input)",
          "aValue": 0.75,
          "bValue": 0.75,
          "unit": "usd",
          "aDisplay": "$0.75",
          "bDisplay": "$0.75",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "gateway-vs-direct-gemini-3-8-flash",
          "aContext": "Gemini 3.8 Flash · list price, snapshot 2026-10-06",
          "bContext": "first-party list price · Gemini 3.8 Flash · list price, snapshot 2026-10-06"
        },
        {
          "metric": "Gemini 3.8 Flash: OpenRouter vs Google list price (Output)",
          "aValue": 3.75,
          "bValue": 3.75,
          "unit": "usd",
          "aDisplay": "$3.75",
          "bDisplay": "$3.75",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "gateway-vs-direct-gemini-3-8-flash",
          "aContext": "Gemini 3.8 Flash · list price, snapshot 2026-10-06",
          "bContext": "first-party list price · Gemini 3.8 Flash · list price, snapshot 2026-10-06"
        },
        {
          "metric": "Gemini 3.5 Flash: OpenRouter vs Google list price (Input)",
          "aValue": 1.5,
          "bValue": 1.5,
          "unit": "usd",
          "aDisplay": "$1.50",
          "bDisplay": "$1.50",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "gateway-vs-direct-gemini-3-5-flash",
          "aContext": "Gemini 3.5 Flash · list price, snapshot 2026-10-06",
          "bContext": "first-party list price · Gemini 3.5 Flash · list price, snapshot 2026-10-06"
        },
        {
          "metric": "Gemini 3.5 Flash: OpenRouter vs Google list price (Output)",
          "aValue": 9,
          "bValue": 9,
          "unit": "usd",
          "aDisplay": "$9.00",
          "bDisplay": "$9.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "gateway-vs-direct-gemini-3-5-flash",
          "aContext": "Gemini 3.5 Flash · list price, snapshot 2026-10-06",
          "bContext": "first-party list price · Gemini 3.5 Flash · list price, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "openrouter-vs-openai",
      "a": "openrouter",
      "b": "openai",
      "title": "OpenRouter vs OpenAI",
      "seoTitle": "OpenRouter vs OpenAI: price per model",
      "description": "OpenRouter vs OpenAI: 2 list prices per million tokens for the same models, as reported by OpenRouter’s public API, with the gap stated.",
      "verdict": "OpenRouter and OpenAI share 2 reported list prices from one study. No row names a winner: a reported list price has no interval, so each gap is stated, not ranked. Prices change often; the snapshot date is in each row’s context. The rows are 2 ties; each row says why.",
      "rows": [
        {
          "metric": "GPT-6 Luna: OpenRouter vs OpenAI list price (Input)",
          "aValue": 0.1,
          "bValue": 0.1,
          "unit": "usd",
          "aDisplay": "$0.10",
          "bDisplay": "$0.10",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "gateway-vs-direct-gpt-6-luna",
          "aContext": "GPT-6 Luna · list price, snapshot 2026-10-06",
          "bContext": "first-party list price · GPT-6 Luna · list price, snapshot 2026-10-06"
        },
        {
          "metric": "GPT-6 Luna: OpenRouter vs OpenAI list price (Output)",
          "aValue": 0.5,
          "bValue": 0.5,
          "unit": "usd",
          "aDisplay": "$0.50",
          "bDisplay": "$0.50",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "gateway-vs-direct-gpt-6-luna",
          "aContext": "GPT-6 Luna · list price, snapshot 2026-10-06",
          "bContext": "first-party list price · GPT-6 Luna · list price, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "groq-vs-together",
      "a": "groq",
      "b": "together",
      "title": "Groq vs Together AI",
      "seoTitle": "Groq vs Together AI: price per model",
      "description": "Groq vs Together AI: 4 list prices per million tokens for the same models, as reported by OpenRouter’s public API, with the gap stated.",
      "verdict": "Groq and Together AI share 4 reported list prices from one study. No row names a winner: a reported list price has no interval, so each gap is stated, not ranked. Prices change often; the snapshot date is in each row’s context. The rows are 2 ties and 2 unclear; each row says why.",
      "rows": [
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Input)",
          "aValue": 0.15,
          "bValue": 0.15,
          "unit": "usd",
          "aDisplay": "$0.15",
          "bDisplay": "$0.15",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Output)",
          "aValue": 0.6,
          "bValue": 0.6,
          "unit": "usd",
          "aDisplay": "$0.60",
          "bDisplay": "$0.60",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Llama 3.3 70B Instruct: price per million tokens by provider (Input)",
          "aValue": 0.59,
          "bValue": 1.04,
          "unit": "usd",
          "aDisplay": "$0.59",
          "bDisplay": "$1.04",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.59 vs $1.04, 1.8x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-llama-3-3-70b-instruct",
          "aContext": "Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Llama 3.3 70B Instruct: price per million tokens by provider (Output)",
          "aValue": 0.79,
          "bValue": 1.04,
          "unit": "usd",
          "aDisplay": "$0.79",
          "bDisplay": "$1.04",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.79 vs $1.04, 1.3x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-llama-3-3-70b-instruct",
          "aContext": "Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "groq-vs-cerebras",
      "a": "groq",
      "b": "cerebras",
      "title": "Groq vs Cerebras",
      "seoTitle": "Groq vs Cerebras: price per model",
      "description": "Groq vs Cerebras: 3 list prices per million tokens for the same models, as reported by OpenRouter’s public API, with the gap stated.",
      "verdict": "Groq and Cerebras share 3 reported list prices from one study. No row names a winner: a reported list price has no interval, so each gap is stated, not ranked. Prices change often; the snapshot date is in each row’s context. The rows are 3 unclear; each row says why.",
      "rows": [
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Input)",
          "aValue": 0.15,
          "bValue": 0.35,
          "unit": "usd",
          "aDisplay": "$0.15",
          "bDisplay": "$0.35",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.15 vs $0.35, 2.3x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp16 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Output)",
          "aValue": 0.6,
          "bValue": 0.75,
          "unit": "usd",
          "aDisplay": "$0.60",
          "bDisplay": "$0.75",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.60 vs $0.75, 1.3x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp16 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Cache read)",
          "aValue": 0.075,
          "bValue": 0.35,
          "unit": "usd",
          "aDisplay": "$0.075",
          "bDisplay": "$0.35",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.075 vs $0.35, 4.7x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp16 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "together-vs-fireworks",
      "a": "together",
      "b": "fireworks",
      "title": "Together AI vs Fireworks AI",
      "seoTitle": "Together AI vs Fireworks AI: price per model",
      "description": "Together AI vs Fireworks AI: 6 list prices per million tokens for the same models, as reported by OpenRouter’s public API, with the gap stated.",
      "verdict": "Together AI and Fireworks AI share 6 reported list prices from one study. No row names a winner: a reported list price has no interval, so each gap is stated, not ranked. Prices change often; the snapshot date is in each row’s context. The rows are 3 ties and 3 unclear; each row says why.",
      "rows": [
        {
          "metric": "Kimi K3: price per million tokens by provider (Input)",
          "aValue": 2.7,
          "bValue": 3,
          "unit": "usd",
          "aDisplay": "$2.70",
          "bDisplay": "$3.00",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($2.70 vs $3.00, 1.1x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-kimi-k3",
          "aContext": "Kimi K3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Kimi K3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Kimi K3: price per million tokens by provider (Output)",
          "aValue": 13.5,
          "bValue": 15,
          "unit": "usd",
          "aDisplay": "$13.50",
          "bDisplay": "$15.00",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($13.50 vs $15.00, 1.1x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-kimi-k3",
          "aContext": "Kimi K3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Kimi K3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Kimi K3: price per million tokens by provider (Cache read)",
          "aValue": 0.27,
          "bValue": 0.3,
          "unit": "usd",
          "aDisplay": "$0.27",
          "bDisplay": "$0.30",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.27 vs $0.30, 1.1x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-kimi-k3",
          "aContext": "Kimi K3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Kimi K3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "GLM 5.3: price per million tokens by provider (Input)",
          "aValue": 1.4,
          "bValue": 1.4,
          "unit": "usd",
          "aDisplay": "$1.40",
          "bDisplay": "$1.40",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "aContext": "GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "GLM 5.3: price per million tokens by provider (Output)",
          "aValue": 4.4,
          "bValue": 4.4,
          "unit": "usd",
          "aDisplay": "$4.40",
          "bDisplay": "$4.40",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "aContext": "GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "GLM 5.3: price per million tokens by provider (Cache read)",
          "aValue": 0.26,
          "bValue": 0.26,
          "unit": "usd",
          "aDisplay": "$0.26",
          "bDisplay": "$0.26",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "aContext": "GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "amazon-bedrock-vs-google-vertex",
      "a": "amazon-bedrock",
      "b": "google-vertex",
      "title": "Amazon Bedrock vs Google Vertex AI",
      "seoTitle": "Amazon Bedrock vs Google Vertex AI: price per model",
      "description": "Amazon Bedrock vs Google Vertex AI: 23 list prices per million tokens for the same models, as reported by OpenRouter’s public API, with the gap stated.",
      "verdict": "Amazon Bedrock and Google Vertex AI share 23 reported list prices from one study. No row names a winner: a reported list price has no interval, so each gap is stated, not ranked. Prices change often; the snapshot date is in each row’s context. The rows are 21 ties and 2 unclear; each row says why.",
      "rows": [
        {
          "metric": "Claude Haiku 4.5: price per million tokens by provider (Input)",
          "aValue": 1,
          "bValue": 1,
          "unit": "usd",
          "aDisplay": "$1.00",
          "bDisplay": "$1.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-haiku-4-5",
          "aContext": "Claude Haiku 4.5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Haiku 4.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Haiku 4.5: price per million tokens by provider (Output)",
          "aValue": 5,
          "bValue": 5,
          "unit": "usd",
          "aDisplay": "$5.00",
          "bDisplay": "$5.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-haiku-4-5",
          "aContext": "Claude Haiku 4.5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Haiku 4.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Haiku 4.5: price per million tokens by provider (Cache read)",
          "aValue": 0.1,
          "bValue": 0.1,
          "unit": "usd",
          "aDisplay": "$0.10",
          "bDisplay": "$0.10",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-haiku-4-5",
          "aContext": "Claude Haiku 4.5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Haiku 4.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Sonnet 5: price per million tokens by provider (Input)",
          "aValue": 2,
          "bValue": 2,
          "unit": "usd",
          "aDisplay": "$2.00",
          "bDisplay": "$2.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-sonnet-5",
          "aContext": "Claude Sonnet 5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Sonnet 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Sonnet 5: price per million tokens by provider (Output)",
          "aValue": 10,
          "bValue": 10,
          "unit": "usd",
          "aDisplay": "$10.00",
          "bDisplay": "$10.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-sonnet-5",
          "aContext": "Claude Sonnet 5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Sonnet 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Sonnet 5: price per million tokens by provider (Cache read)",
          "aValue": 0.2,
          "bValue": 0.2,
          "unit": "usd",
          "aDisplay": "$0.20",
          "bDisplay": "$0.20",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-sonnet-5",
          "aContext": "Claude Sonnet 5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Sonnet 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Sonnet 5.5: price per million tokens by provider (Input)",
          "aValue": 2,
          "bValue": 2,
          "unit": "usd",
          "aDisplay": "$2.00",
          "bDisplay": "$2.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-sonnet-5-5",
          "aContext": "Claude Sonnet 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Sonnet 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Sonnet 5.5: price per million tokens by provider (Output)",
          "aValue": 10,
          "bValue": 10,
          "unit": "usd",
          "aDisplay": "$10.00",
          "bDisplay": "$10.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-sonnet-5-5",
          "aContext": "Claude Sonnet 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Sonnet 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Sonnet 5.5: price per million tokens by provider (Cache read)",
          "aValue": 0.2,
          "bValue": 0.2,
          "unit": "usd",
          "aDisplay": "$0.20",
          "bDisplay": "$0.20",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-sonnet-5-5",
          "aContext": "Claude Sonnet 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Sonnet 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Opus 4.8: price per million tokens by provider (Input)",
          "aValue": 5,
          "bValue": 5,
          "unit": "usd",
          "aDisplay": "$5.00",
          "bDisplay": "$5.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-4-8",
          "aContext": "Claude Opus 4.8 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Opus 4.8 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Opus 4.8: price per million tokens by provider (Output)",
          "aValue": 25,
          "bValue": 25,
          "unit": "usd",
          "aDisplay": "$25.00",
          "bDisplay": "$25.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-4-8",
          "aContext": "Claude Opus 4.8 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Opus 4.8 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Opus 4.8: price per million tokens by provider (Cache read)",
          "aValue": 0.5,
          "bValue": 0.5,
          "unit": "usd",
          "aDisplay": "$0.50",
          "bDisplay": "$0.50",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-4-8",
          "aContext": "Claude Opus 4.8 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Opus 4.8 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Opus 5: price per million tokens by provider (Input)",
          "aValue": 5,
          "bValue": 5,
          "unit": "usd",
          "aDisplay": "$5.00",
          "bDisplay": "$5.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-5",
          "aContext": "Claude Opus 5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Opus 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Opus 5: price per million tokens by provider (Output)",
          "aValue": 25,
          "bValue": 25,
          "unit": "usd",
          "aDisplay": "$25.00",
          "bDisplay": "$25.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-5",
          "aContext": "Claude Opus 5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Opus 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Opus 5: price per million tokens by provider (Cache read)",
          "aValue": 0.5,
          "bValue": 0.5,
          "unit": "usd",
          "aDisplay": "$0.50",
          "bDisplay": "$0.50",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-5",
          "aContext": "Claude Opus 5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Opus 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Opus 5.5: price per million tokens by provider (Input)",
          "aValue": 4,
          "bValue": 4,
          "unit": "usd",
          "aDisplay": "$4.00",
          "bDisplay": "$4.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-5-5",
          "aContext": "Claude Opus 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Opus 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Opus 5.5: price per million tokens by provider (Output)",
          "aValue": 20,
          "bValue": 20,
          "unit": "usd",
          "aDisplay": "$20.00",
          "bDisplay": "$20.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-5-5",
          "aContext": "Claude Opus 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Opus 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Opus 5.5: price per million tokens by provider (Cache read)",
          "aValue": 0.2,
          "bValue": 0.2,
          "unit": "usd",
          "aDisplay": "$0.20",
          "bDisplay": "$0.20",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-5-5",
          "aContext": "Claude Opus 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Opus 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Fable 5.1: price per million tokens by provider (Input)",
          "aValue": 10,
          "bValue": 10,
          "unit": "usd",
          "aDisplay": "$10.00",
          "bDisplay": "$10.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-fable-5-1",
          "aContext": "Claude Fable 5.1 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Fable 5.1 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Fable 5.1: price per million tokens by provider (Output)",
          "aValue": 50,
          "bValue": 50,
          "unit": "usd",
          "aDisplay": "$50.00",
          "bDisplay": "$50.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-fable-5-1",
          "aContext": "Claude Fable 5.1 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Fable 5.1 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Fable 5.1: price per million tokens by provider (Cache read)",
          "aValue": 0.25,
          "bValue": 0.25,
          "unit": "usd",
          "aDisplay": "$0.25",
          "bDisplay": "$0.25",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-fable-5-1",
          "aContext": "Claude Fable 5.1 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Fable 5.1 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Input)",
          "aValue": 0.15,
          "bValue": 0.09,
          "unit": "usd",
          "aDisplay": "$0.15",
          "bDisplay": "$0.090",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.15 vs $0.090, 1.7x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Output)",
          "aValue": 0.6,
          "bValue": 0.36,
          "unit": "usd",
          "aDisplay": "$0.60",
          "bDisplay": "$0.36",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.60 vs $0.36, 1.7x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "amazon-bedrock-vs-azure",
      "a": "amazon-bedrock",
      "b": "azure",
      "title": "Amazon Bedrock vs Azure",
      "seoTitle": "Amazon Bedrock vs Azure: price per model",
      "description": "Amazon Bedrock vs Azure: 21 list prices per million tokens for the same models, as reported by OpenRouter’s public API, with the gap stated.",
      "verdict": "Amazon Bedrock and Azure share 21 reported list prices from one study. No row names a winner: a reported list price has no interval, so each gap is stated, not ranked. Prices change often; the snapshot date is in each row’s context. The rows are 21 ties; each row says why.",
      "rows": [
        {
          "metric": "Claude Haiku 4.5: price per million tokens by provider (Input)",
          "aValue": 1,
          "bValue": 1,
          "unit": "usd",
          "aDisplay": "$1.00",
          "bDisplay": "$1.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-haiku-4-5",
          "aContext": "Claude Haiku 4.5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Haiku 4.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Haiku 4.5: price per million tokens by provider (Output)",
          "aValue": 5,
          "bValue": 5,
          "unit": "usd",
          "aDisplay": "$5.00",
          "bDisplay": "$5.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-haiku-4-5",
          "aContext": "Claude Haiku 4.5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Haiku 4.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Haiku 4.5: price per million tokens by provider (Cache read)",
          "aValue": 0.1,
          "bValue": 0.1,
          "unit": "usd",
          "aDisplay": "$0.10",
          "bDisplay": "$0.10",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-haiku-4-5",
          "aContext": "Claude Haiku 4.5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Haiku 4.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Sonnet 5: price per million tokens by provider (Input)",
          "aValue": 2,
          "bValue": 2,
          "unit": "usd",
          "aDisplay": "$2.00",
          "bDisplay": "$2.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-sonnet-5",
          "aContext": "Claude Sonnet 5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Sonnet 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Sonnet 5: price per million tokens by provider (Output)",
          "aValue": 10,
          "bValue": 10,
          "unit": "usd",
          "aDisplay": "$10.00",
          "bDisplay": "$10.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-sonnet-5",
          "aContext": "Claude Sonnet 5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Sonnet 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Sonnet 5: price per million tokens by provider (Cache read)",
          "aValue": 0.2,
          "bValue": 0.2,
          "unit": "usd",
          "aDisplay": "$0.20",
          "bDisplay": "$0.20",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-sonnet-5",
          "aContext": "Claude Sonnet 5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Sonnet 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Sonnet 5.5: price per million tokens by provider (Input)",
          "aValue": 2,
          "bValue": 2,
          "unit": "usd",
          "aDisplay": "$2.00",
          "bDisplay": "$2.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-sonnet-5-5",
          "aContext": "Claude Sonnet 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Sonnet 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Sonnet 5.5: price per million tokens by provider (Output)",
          "aValue": 10,
          "bValue": 10,
          "unit": "usd",
          "aDisplay": "$10.00",
          "bDisplay": "$10.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-sonnet-5-5",
          "aContext": "Claude Sonnet 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Sonnet 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Sonnet 5.5: price per million tokens by provider (Cache read)",
          "aValue": 0.2,
          "bValue": 0.2,
          "unit": "usd",
          "aDisplay": "$0.20",
          "bDisplay": "$0.20",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-sonnet-5-5",
          "aContext": "Claude Sonnet 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Sonnet 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Opus 4.8: price per million tokens by provider (Input)",
          "aValue": 5,
          "bValue": 5,
          "unit": "usd",
          "aDisplay": "$5.00",
          "bDisplay": "$5.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-4-8",
          "aContext": "Claude Opus 4.8 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Opus 4.8 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Opus 4.8: price per million tokens by provider (Output)",
          "aValue": 25,
          "bValue": 25,
          "unit": "usd",
          "aDisplay": "$25.00",
          "bDisplay": "$25.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-4-8",
          "aContext": "Claude Opus 4.8 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Opus 4.8 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Opus 4.8: price per million tokens by provider (Cache read)",
          "aValue": 0.5,
          "bValue": 0.5,
          "unit": "usd",
          "aDisplay": "$0.50",
          "bDisplay": "$0.50",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-4-8",
          "aContext": "Claude Opus 4.8 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Opus 4.8 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Opus 5: price per million tokens by provider (Input)",
          "aValue": 5,
          "bValue": 5,
          "unit": "usd",
          "aDisplay": "$5.00",
          "bDisplay": "$5.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-5",
          "aContext": "Claude Opus 5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Opus 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Opus 5: price per million tokens by provider (Output)",
          "aValue": 25,
          "bValue": 25,
          "unit": "usd",
          "aDisplay": "$25.00",
          "bDisplay": "$25.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-5",
          "aContext": "Claude Opus 5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Opus 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Opus 5: price per million tokens by provider (Cache read)",
          "aValue": 0.5,
          "bValue": 0.5,
          "unit": "usd",
          "aDisplay": "$0.50",
          "bDisplay": "$0.50",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-5",
          "aContext": "Claude Opus 5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Opus 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Opus 5.5: price per million tokens by provider (Input)",
          "aValue": 4,
          "bValue": 4,
          "unit": "usd",
          "aDisplay": "$4.00",
          "bDisplay": "$4.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-5-5",
          "aContext": "Claude Opus 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Opus 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Opus 5.5: price per million tokens by provider (Output)",
          "aValue": 20,
          "bValue": 20,
          "unit": "usd",
          "aDisplay": "$20.00",
          "bDisplay": "$20.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-5-5",
          "aContext": "Claude Opus 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Opus 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Opus 5.5: price per million tokens by provider (Cache read)",
          "aValue": 0.2,
          "bValue": 0.2,
          "unit": "usd",
          "aDisplay": "$0.20",
          "bDisplay": "$0.20",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-5-5",
          "aContext": "Claude Opus 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Opus 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Fable 5.1: price per million tokens by provider (Input)",
          "aValue": 10,
          "bValue": 10,
          "unit": "usd",
          "aDisplay": "$10.00",
          "bDisplay": "$10.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-fable-5-1",
          "aContext": "Claude Fable 5.1 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Fable 5.1 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Fable 5.1: price per million tokens by provider (Output)",
          "aValue": 50,
          "bValue": 50,
          "unit": "usd",
          "aDisplay": "$50.00",
          "bDisplay": "$50.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-fable-5-1",
          "aContext": "Claude Fable 5.1 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Fable 5.1 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Fable 5.1: price per million tokens by provider (Cache read)",
          "aValue": 0.25,
          "bValue": 0.25,
          "unit": "usd",
          "aDisplay": "$0.25",
          "bDisplay": "$0.25",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-fable-5-1",
          "aContext": "Claude Fable 5.1 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Fable 5.1 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "anthropic-vs-amazon-bedrock",
      "a": "anthropic",
      "b": "amazon-bedrock",
      "title": "Anthropic vs Amazon Bedrock",
      "seoTitle": "Anthropic vs Amazon Bedrock: price per model",
      "description": "Anthropic vs Amazon Bedrock: 21 list prices per million tokens for the same models, as reported by OpenRouter’s public API, with the gap stated.",
      "verdict": "Anthropic and Amazon Bedrock share 21 reported list prices from one study. No row names a winner: a reported list price has no interval, so each gap is stated, not ranked. Prices change often; the snapshot date is in each row’s context. The rows are 21 ties; each row says why.",
      "rows": [
        {
          "metric": "Claude Haiku 4.5: price per million tokens by provider (Input)",
          "aValue": 1,
          "bValue": 1,
          "unit": "usd",
          "aDisplay": "$1.00",
          "bDisplay": "$1.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-haiku-4-5",
          "aContext": "Claude Haiku 4.5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Haiku 4.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Haiku 4.5: price per million tokens by provider (Output)",
          "aValue": 5,
          "bValue": 5,
          "unit": "usd",
          "aDisplay": "$5.00",
          "bDisplay": "$5.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-haiku-4-5",
          "aContext": "Claude Haiku 4.5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Haiku 4.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Haiku 4.5: price per million tokens by provider (Cache read)",
          "aValue": 0.1,
          "bValue": 0.1,
          "unit": "usd",
          "aDisplay": "$0.10",
          "bDisplay": "$0.10",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-haiku-4-5",
          "aContext": "Claude Haiku 4.5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Haiku 4.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Sonnet 5: price per million tokens by provider (Input)",
          "aValue": 2,
          "bValue": 2,
          "unit": "usd",
          "aDisplay": "$2.00",
          "bDisplay": "$2.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-sonnet-5",
          "aContext": "Claude Sonnet 5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Sonnet 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Sonnet 5: price per million tokens by provider (Output)",
          "aValue": 10,
          "bValue": 10,
          "unit": "usd",
          "aDisplay": "$10.00",
          "bDisplay": "$10.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-sonnet-5",
          "aContext": "Claude Sonnet 5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Sonnet 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Sonnet 5: price per million tokens by provider (Cache read)",
          "aValue": 0.2,
          "bValue": 0.2,
          "unit": "usd",
          "aDisplay": "$0.20",
          "bDisplay": "$0.20",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-sonnet-5",
          "aContext": "Claude Sonnet 5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Sonnet 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Sonnet 5.5: price per million tokens by provider (Input)",
          "aValue": 2,
          "bValue": 2,
          "unit": "usd",
          "aDisplay": "$2.00",
          "bDisplay": "$2.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-sonnet-5-5",
          "aContext": "Claude Sonnet 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Sonnet 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Sonnet 5.5: price per million tokens by provider (Output)",
          "aValue": 10,
          "bValue": 10,
          "unit": "usd",
          "aDisplay": "$10.00",
          "bDisplay": "$10.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-sonnet-5-5",
          "aContext": "Claude Sonnet 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Sonnet 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Sonnet 5.5: price per million tokens by provider (Cache read)",
          "aValue": 0.2,
          "bValue": 0.2,
          "unit": "usd",
          "aDisplay": "$0.20",
          "bDisplay": "$0.20",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-sonnet-5-5",
          "aContext": "Claude Sonnet 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Sonnet 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Opus 4.8: price per million tokens by provider (Input)",
          "aValue": 5,
          "bValue": 5,
          "unit": "usd",
          "aDisplay": "$5.00",
          "bDisplay": "$5.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-4-8",
          "aContext": "Claude Opus 4.8 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Opus 4.8 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Opus 4.8: price per million tokens by provider (Output)",
          "aValue": 25,
          "bValue": 25,
          "unit": "usd",
          "aDisplay": "$25.00",
          "bDisplay": "$25.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-4-8",
          "aContext": "Claude Opus 4.8 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Opus 4.8 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Opus 4.8: price per million tokens by provider (Cache read)",
          "aValue": 0.5,
          "bValue": 0.5,
          "unit": "usd",
          "aDisplay": "$0.50",
          "bDisplay": "$0.50",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-4-8",
          "aContext": "Claude Opus 4.8 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Opus 4.8 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Opus 5: price per million tokens by provider (Input)",
          "aValue": 5,
          "bValue": 5,
          "unit": "usd",
          "aDisplay": "$5.00",
          "bDisplay": "$5.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-5",
          "aContext": "Claude Opus 5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Opus 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Opus 5: price per million tokens by provider (Output)",
          "aValue": 25,
          "bValue": 25,
          "unit": "usd",
          "aDisplay": "$25.00",
          "bDisplay": "$25.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-5",
          "aContext": "Claude Opus 5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Opus 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Opus 5: price per million tokens by provider (Cache read)",
          "aValue": 0.5,
          "bValue": 0.5,
          "unit": "usd",
          "aDisplay": "$0.50",
          "bDisplay": "$0.50",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-5",
          "aContext": "Claude Opus 5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Opus 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Opus 5.5: price per million tokens by provider (Input)",
          "aValue": 4,
          "bValue": 4,
          "unit": "usd",
          "aDisplay": "$4.00",
          "bDisplay": "$4.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-5-5",
          "aContext": "Claude Opus 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Opus 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Opus 5.5: price per million tokens by provider (Output)",
          "aValue": 20,
          "bValue": 20,
          "unit": "usd",
          "aDisplay": "$20.00",
          "bDisplay": "$20.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-5-5",
          "aContext": "Claude Opus 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Opus 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Opus 5.5: price per million tokens by provider (Cache read)",
          "aValue": 0.2,
          "bValue": 0.2,
          "unit": "usd",
          "aDisplay": "$0.20",
          "bDisplay": "$0.20",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-5-5",
          "aContext": "Claude Opus 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Opus 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Fable 5.1: price per million tokens by provider (Input)",
          "aValue": 10,
          "bValue": 10,
          "unit": "usd",
          "aDisplay": "$10.00",
          "bDisplay": "$10.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-fable-5-1",
          "aContext": "Claude Fable 5.1 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Fable 5.1 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Fable 5.1: price per million tokens by provider (Output)",
          "aValue": 50,
          "bValue": 50,
          "unit": "usd",
          "aDisplay": "$50.00",
          "bDisplay": "$50.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-fable-5-1",
          "aContext": "Claude Fable 5.1 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Fable 5.1 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Fable 5.1: price per million tokens by provider (Cache read)",
          "aValue": 0.25,
          "bValue": 0.25,
          "unit": "usd",
          "aDisplay": "$0.25",
          "bDisplay": "$0.25",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-fable-5-1",
          "aContext": "Claude Fable 5.1 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Fable 5.1 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "anthropic-vs-azure",
      "a": "anthropic",
      "b": "azure",
      "title": "Anthropic vs Azure",
      "seoTitle": "Anthropic vs Azure: price per model",
      "description": "Anthropic vs Azure: 21 list prices per million tokens for the same models, as reported by OpenRouter’s public API, with the gap stated.",
      "verdict": "Anthropic and Azure share 21 reported list prices from one study. No row names a winner: a reported list price has no interval, so each gap is stated, not ranked. Prices change often; the snapshot date is in each row’s context. The rows are 21 ties; each row says why.",
      "rows": [
        {
          "metric": "Claude Haiku 4.5: price per million tokens by provider (Input)",
          "aValue": 1,
          "bValue": 1,
          "unit": "usd",
          "aDisplay": "$1.00",
          "bDisplay": "$1.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-haiku-4-5",
          "aContext": "Claude Haiku 4.5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Haiku 4.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Haiku 4.5: price per million tokens by provider (Output)",
          "aValue": 5,
          "bValue": 5,
          "unit": "usd",
          "aDisplay": "$5.00",
          "bDisplay": "$5.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-haiku-4-5",
          "aContext": "Claude Haiku 4.5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Haiku 4.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Haiku 4.5: price per million tokens by provider (Cache read)",
          "aValue": 0.1,
          "bValue": 0.1,
          "unit": "usd",
          "aDisplay": "$0.10",
          "bDisplay": "$0.10",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-haiku-4-5",
          "aContext": "Claude Haiku 4.5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Haiku 4.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Sonnet 5: price per million tokens by provider (Input)",
          "aValue": 2,
          "bValue": 2,
          "unit": "usd",
          "aDisplay": "$2.00",
          "bDisplay": "$2.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-sonnet-5",
          "aContext": "Claude Sonnet 5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Sonnet 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Sonnet 5: price per million tokens by provider (Output)",
          "aValue": 10,
          "bValue": 10,
          "unit": "usd",
          "aDisplay": "$10.00",
          "bDisplay": "$10.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-sonnet-5",
          "aContext": "Claude Sonnet 5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Sonnet 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Sonnet 5: price per million tokens by provider (Cache read)",
          "aValue": 0.2,
          "bValue": 0.2,
          "unit": "usd",
          "aDisplay": "$0.20",
          "bDisplay": "$0.20",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-sonnet-5",
          "aContext": "Claude Sonnet 5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Sonnet 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Sonnet 5.5: price per million tokens by provider (Input)",
          "aValue": 2,
          "bValue": 2,
          "unit": "usd",
          "aDisplay": "$2.00",
          "bDisplay": "$2.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-sonnet-5-5",
          "aContext": "Claude Sonnet 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Sonnet 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Sonnet 5.5: price per million tokens by provider (Output)",
          "aValue": 10,
          "bValue": 10,
          "unit": "usd",
          "aDisplay": "$10.00",
          "bDisplay": "$10.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-sonnet-5-5",
          "aContext": "Claude Sonnet 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Sonnet 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Sonnet 5.5: price per million tokens by provider (Cache read)",
          "aValue": 0.2,
          "bValue": 0.2,
          "unit": "usd",
          "aDisplay": "$0.20",
          "bDisplay": "$0.20",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-sonnet-5-5",
          "aContext": "Claude Sonnet 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Sonnet 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Opus 4.8: price per million tokens by provider (Input)",
          "aValue": 5,
          "bValue": 5,
          "unit": "usd",
          "aDisplay": "$5.00",
          "bDisplay": "$5.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-4-8",
          "aContext": "Claude Opus 4.8 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Opus 4.8 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Opus 4.8: price per million tokens by provider (Output)",
          "aValue": 25,
          "bValue": 25,
          "unit": "usd",
          "aDisplay": "$25.00",
          "bDisplay": "$25.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-4-8",
          "aContext": "Claude Opus 4.8 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Opus 4.8 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Opus 4.8: price per million tokens by provider (Cache read)",
          "aValue": 0.5,
          "bValue": 0.5,
          "unit": "usd",
          "aDisplay": "$0.50",
          "bDisplay": "$0.50",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-4-8",
          "aContext": "Claude Opus 4.8 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Opus 4.8 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Opus 5: price per million tokens by provider (Input)",
          "aValue": 5,
          "bValue": 5,
          "unit": "usd",
          "aDisplay": "$5.00",
          "bDisplay": "$5.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-5",
          "aContext": "Claude Opus 5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Opus 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Opus 5: price per million tokens by provider (Output)",
          "aValue": 25,
          "bValue": 25,
          "unit": "usd",
          "aDisplay": "$25.00",
          "bDisplay": "$25.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-5",
          "aContext": "Claude Opus 5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Opus 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Opus 5: price per million tokens by provider (Cache read)",
          "aValue": 0.5,
          "bValue": 0.5,
          "unit": "usd",
          "aDisplay": "$0.50",
          "bDisplay": "$0.50",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-5",
          "aContext": "Claude Opus 5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Opus 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Opus 5.5: price per million tokens by provider (Input)",
          "aValue": 4,
          "bValue": 4,
          "unit": "usd",
          "aDisplay": "$4.00",
          "bDisplay": "$4.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-5-5",
          "aContext": "Claude Opus 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Opus 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Opus 5.5: price per million tokens by provider (Output)",
          "aValue": 20,
          "bValue": 20,
          "unit": "usd",
          "aDisplay": "$20.00",
          "bDisplay": "$20.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-5-5",
          "aContext": "Claude Opus 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Opus 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Opus 5.5: price per million tokens by provider (Cache read)",
          "aValue": 0.2,
          "bValue": 0.2,
          "unit": "usd",
          "aDisplay": "$0.20",
          "bDisplay": "$0.20",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-5-5",
          "aContext": "Claude Opus 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Opus 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Fable 5.1: price per million tokens by provider (Input)",
          "aValue": 10,
          "bValue": 10,
          "unit": "usd",
          "aDisplay": "$10.00",
          "bDisplay": "$10.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-fable-5-1",
          "aContext": "Claude Fable 5.1 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Fable 5.1 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Fable 5.1: price per million tokens by provider (Output)",
          "aValue": 50,
          "bValue": 50,
          "unit": "usd",
          "aDisplay": "$50.00",
          "bDisplay": "$50.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-fable-5-1",
          "aContext": "Claude Fable 5.1 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Fable 5.1 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Fable 5.1: price per million tokens by provider (Cache read)",
          "aValue": 0.25,
          "bValue": 0.25,
          "unit": "usd",
          "aDisplay": "$0.25",
          "bDisplay": "$0.25",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-fable-5-1",
          "aContext": "Claude Fable 5.1 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Fable 5.1 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "anthropic-vs-google-vertex",
      "a": "anthropic",
      "b": "google-vertex",
      "title": "Anthropic vs Google Vertex AI",
      "seoTitle": "Anthropic vs Google Vertex AI: price per model",
      "description": "Anthropic vs Google Vertex AI: 21 list prices per million tokens for the same models, as reported by OpenRouter’s public API, with the gap stated.",
      "verdict": "Anthropic and Google Vertex AI share 21 reported list prices from one study. No row names a winner: a reported list price has no interval, so each gap is stated, not ranked. Prices change often; the snapshot date is in each row’s context. The rows are 21 ties; each row says why.",
      "rows": [
        {
          "metric": "Claude Haiku 4.5: price per million tokens by provider (Input)",
          "aValue": 1,
          "bValue": 1,
          "unit": "usd",
          "aDisplay": "$1.00",
          "bDisplay": "$1.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-haiku-4-5",
          "aContext": "Claude Haiku 4.5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Haiku 4.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Haiku 4.5: price per million tokens by provider (Output)",
          "aValue": 5,
          "bValue": 5,
          "unit": "usd",
          "aDisplay": "$5.00",
          "bDisplay": "$5.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-haiku-4-5",
          "aContext": "Claude Haiku 4.5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Haiku 4.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Haiku 4.5: price per million tokens by provider (Cache read)",
          "aValue": 0.1,
          "bValue": 0.1,
          "unit": "usd",
          "aDisplay": "$0.10",
          "bDisplay": "$0.10",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-haiku-4-5",
          "aContext": "Claude Haiku 4.5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Haiku 4.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Sonnet 5: price per million tokens by provider (Input)",
          "aValue": 2,
          "bValue": 2,
          "unit": "usd",
          "aDisplay": "$2.00",
          "bDisplay": "$2.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-sonnet-5",
          "aContext": "Claude Sonnet 5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Sonnet 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Sonnet 5: price per million tokens by provider (Output)",
          "aValue": 10,
          "bValue": 10,
          "unit": "usd",
          "aDisplay": "$10.00",
          "bDisplay": "$10.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-sonnet-5",
          "aContext": "Claude Sonnet 5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Sonnet 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Sonnet 5: price per million tokens by provider (Cache read)",
          "aValue": 0.2,
          "bValue": 0.2,
          "unit": "usd",
          "aDisplay": "$0.20",
          "bDisplay": "$0.20",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-sonnet-5",
          "aContext": "Claude Sonnet 5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Sonnet 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Sonnet 5.5: price per million tokens by provider (Input)",
          "aValue": 2,
          "bValue": 2,
          "unit": "usd",
          "aDisplay": "$2.00",
          "bDisplay": "$2.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-sonnet-5-5",
          "aContext": "Claude Sonnet 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Sonnet 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Sonnet 5.5: price per million tokens by provider (Output)",
          "aValue": 10,
          "bValue": 10,
          "unit": "usd",
          "aDisplay": "$10.00",
          "bDisplay": "$10.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-sonnet-5-5",
          "aContext": "Claude Sonnet 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Sonnet 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Sonnet 5.5: price per million tokens by provider (Cache read)",
          "aValue": 0.2,
          "bValue": 0.2,
          "unit": "usd",
          "aDisplay": "$0.20",
          "bDisplay": "$0.20",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-sonnet-5-5",
          "aContext": "Claude Sonnet 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Sonnet 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Opus 4.8: price per million tokens by provider (Input)",
          "aValue": 5,
          "bValue": 5,
          "unit": "usd",
          "aDisplay": "$5.00",
          "bDisplay": "$5.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-4-8",
          "aContext": "Claude Opus 4.8 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Opus 4.8 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Opus 4.8: price per million tokens by provider (Output)",
          "aValue": 25,
          "bValue": 25,
          "unit": "usd",
          "aDisplay": "$25.00",
          "bDisplay": "$25.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-4-8",
          "aContext": "Claude Opus 4.8 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Opus 4.8 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Opus 4.8: price per million tokens by provider (Cache read)",
          "aValue": 0.5,
          "bValue": 0.5,
          "unit": "usd",
          "aDisplay": "$0.50",
          "bDisplay": "$0.50",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-4-8",
          "aContext": "Claude Opus 4.8 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Opus 4.8 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Opus 5: price per million tokens by provider (Input)",
          "aValue": 5,
          "bValue": 5,
          "unit": "usd",
          "aDisplay": "$5.00",
          "bDisplay": "$5.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-5",
          "aContext": "Claude Opus 5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Opus 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Opus 5: price per million tokens by provider (Output)",
          "aValue": 25,
          "bValue": 25,
          "unit": "usd",
          "aDisplay": "$25.00",
          "bDisplay": "$25.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-5",
          "aContext": "Claude Opus 5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Opus 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Opus 5: price per million tokens by provider (Cache read)",
          "aValue": 0.5,
          "bValue": 0.5,
          "unit": "usd",
          "aDisplay": "$0.50",
          "bDisplay": "$0.50",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-5",
          "aContext": "Claude Opus 5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Opus 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Opus 5.5: price per million tokens by provider (Input)",
          "aValue": 4,
          "bValue": 4,
          "unit": "usd",
          "aDisplay": "$4.00",
          "bDisplay": "$4.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-5-5",
          "aContext": "Claude Opus 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Opus 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Opus 5.5: price per million tokens by provider (Output)",
          "aValue": 20,
          "bValue": 20,
          "unit": "usd",
          "aDisplay": "$20.00",
          "bDisplay": "$20.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-5-5",
          "aContext": "Claude Opus 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Opus 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Opus 5.5: price per million tokens by provider (Cache read)",
          "aValue": 0.2,
          "bValue": 0.2,
          "unit": "usd",
          "aDisplay": "$0.20",
          "bDisplay": "$0.20",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-5-5",
          "aContext": "Claude Opus 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Opus 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Fable 5.1: price per million tokens by provider (Input)",
          "aValue": 10,
          "bValue": 10,
          "unit": "usd",
          "aDisplay": "$10.00",
          "bDisplay": "$10.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-fable-5-1",
          "aContext": "Claude Fable 5.1 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Fable 5.1 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Fable 5.1: price per million tokens by provider (Output)",
          "aValue": 50,
          "bValue": 50,
          "unit": "usd",
          "aDisplay": "$50.00",
          "bDisplay": "$50.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-fable-5-1",
          "aContext": "Claude Fable 5.1 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Fable 5.1 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Fable 5.1: price per million tokens by provider (Cache read)",
          "aValue": 0.25,
          "bValue": 0.25,
          "unit": "usd",
          "aDisplay": "$0.25",
          "bDisplay": "$0.25",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-fable-5-1",
          "aContext": "Claude Fable 5.1 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Fable 5.1 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "google-vertex-vs-azure",
      "a": "google-vertex",
      "b": "azure",
      "title": "Google Vertex AI vs Azure",
      "seoTitle": "Google Vertex AI vs Azure: price per model",
      "description": "Google Vertex AI vs Azure: 21 list prices per million tokens for the same models, as reported by OpenRouter’s public API, with the gap stated.",
      "verdict": "Google Vertex AI and Azure share 21 reported list prices from one study. No row names a winner: a reported list price has no interval, so each gap is stated, not ranked. Prices change often; the snapshot date is in each row’s context. The rows are 21 ties; each row says why.",
      "rows": [
        {
          "metric": "Claude Haiku 4.5: price per million tokens by provider (Input)",
          "aValue": 1,
          "bValue": 1,
          "unit": "usd",
          "aDisplay": "$1.00",
          "bDisplay": "$1.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-haiku-4-5",
          "aContext": "Claude Haiku 4.5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Haiku 4.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Haiku 4.5: price per million tokens by provider (Output)",
          "aValue": 5,
          "bValue": 5,
          "unit": "usd",
          "aDisplay": "$5.00",
          "bDisplay": "$5.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-haiku-4-5",
          "aContext": "Claude Haiku 4.5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Haiku 4.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Haiku 4.5: price per million tokens by provider (Cache read)",
          "aValue": 0.1,
          "bValue": 0.1,
          "unit": "usd",
          "aDisplay": "$0.10",
          "bDisplay": "$0.10",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-haiku-4-5",
          "aContext": "Claude Haiku 4.5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Haiku 4.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Sonnet 5: price per million tokens by provider (Input)",
          "aValue": 2,
          "bValue": 2,
          "unit": "usd",
          "aDisplay": "$2.00",
          "bDisplay": "$2.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-sonnet-5",
          "aContext": "Claude Sonnet 5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Sonnet 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Sonnet 5: price per million tokens by provider (Output)",
          "aValue": 10,
          "bValue": 10,
          "unit": "usd",
          "aDisplay": "$10.00",
          "bDisplay": "$10.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-sonnet-5",
          "aContext": "Claude Sonnet 5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Sonnet 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Sonnet 5: price per million tokens by provider (Cache read)",
          "aValue": 0.2,
          "bValue": 0.2,
          "unit": "usd",
          "aDisplay": "$0.20",
          "bDisplay": "$0.20",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-sonnet-5",
          "aContext": "Claude Sonnet 5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Sonnet 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Sonnet 5.5: price per million tokens by provider (Input)",
          "aValue": 2,
          "bValue": 2,
          "unit": "usd",
          "aDisplay": "$2.00",
          "bDisplay": "$2.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-sonnet-5-5",
          "aContext": "Claude Sonnet 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Sonnet 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Sonnet 5.5: price per million tokens by provider (Output)",
          "aValue": 10,
          "bValue": 10,
          "unit": "usd",
          "aDisplay": "$10.00",
          "bDisplay": "$10.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-sonnet-5-5",
          "aContext": "Claude Sonnet 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Sonnet 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Sonnet 5.5: price per million tokens by provider (Cache read)",
          "aValue": 0.2,
          "bValue": 0.2,
          "unit": "usd",
          "aDisplay": "$0.20",
          "bDisplay": "$0.20",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-sonnet-5-5",
          "aContext": "Claude Sonnet 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Sonnet 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Opus 4.8: price per million tokens by provider (Input)",
          "aValue": 5,
          "bValue": 5,
          "unit": "usd",
          "aDisplay": "$5.00",
          "bDisplay": "$5.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-4-8",
          "aContext": "Claude Opus 4.8 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Opus 4.8 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Opus 4.8: price per million tokens by provider (Output)",
          "aValue": 25,
          "bValue": 25,
          "unit": "usd",
          "aDisplay": "$25.00",
          "bDisplay": "$25.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-4-8",
          "aContext": "Claude Opus 4.8 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Opus 4.8 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Opus 4.8: price per million tokens by provider (Cache read)",
          "aValue": 0.5,
          "bValue": 0.5,
          "unit": "usd",
          "aDisplay": "$0.50",
          "bDisplay": "$0.50",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-4-8",
          "aContext": "Claude Opus 4.8 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Opus 4.8 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Opus 5: price per million tokens by provider (Input)",
          "aValue": 5,
          "bValue": 5,
          "unit": "usd",
          "aDisplay": "$5.00",
          "bDisplay": "$5.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-5",
          "aContext": "Claude Opus 5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Opus 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Opus 5: price per million tokens by provider (Output)",
          "aValue": 25,
          "bValue": 25,
          "unit": "usd",
          "aDisplay": "$25.00",
          "bDisplay": "$25.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-5",
          "aContext": "Claude Opus 5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Opus 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Opus 5: price per million tokens by provider (Cache read)",
          "aValue": 0.5,
          "bValue": 0.5,
          "unit": "usd",
          "aDisplay": "$0.50",
          "bDisplay": "$0.50",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-5",
          "aContext": "Claude Opus 5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Opus 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Opus 5.5: price per million tokens by provider (Input)",
          "aValue": 4,
          "bValue": 4,
          "unit": "usd",
          "aDisplay": "$4.00",
          "bDisplay": "$4.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-5-5",
          "aContext": "Claude Opus 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Opus 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Opus 5.5: price per million tokens by provider (Output)",
          "aValue": 20,
          "bValue": 20,
          "unit": "usd",
          "aDisplay": "$20.00",
          "bDisplay": "$20.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-5-5",
          "aContext": "Claude Opus 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Opus 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Opus 5.5: price per million tokens by provider (Cache read)",
          "aValue": 0.2,
          "bValue": 0.2,
          "unit": "usd",
          "aDisplay": "$0.20",
          "bDisplay": "$0.20",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-5-5",
          "aContext": "Claude Opus 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Opus 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Fable 5.1: price per million tokens by provider (Input)",
          "aValue": 10,
          "bValue": 10,
          "unit": "usd",
          "aDisplay": "$10.00",
          "bDisplay": "$10.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-fable-5-1",
          "aContext": "Claude Fable 5.1 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Fable 5.1 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Fable 5.1: price per million tokens by provider (Output)",
          "aValue": 50,
          "bValue": 50,
          "unit": "usd",
          "aDisplay": "$50.00",
          "bDisplay": "$50.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-fable-5-1",
          "aContext": "Claude Fable 5.1 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Fable 5.1 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Fable 5.1: price per million tokens by provider (Cache read)",
          "aValue": 0.25,
          "bValue": 0.25,
          "unit": "usd",
          "aDisplay": "$0.25",
          "bDisplay": "$0.25",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-fable-5-1",
          "aContext": "Claude Fable 5.1 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Fable 5.1 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "deepinfra-vs-parasail",
      "a": "deepinfra",
      "b": "parasail",
      "title": "DeepInfra vs Parasail",
      "seoTitle": "DeepInfra vs Parasail: price per model",
      "description": "DeepInfra vs Parasail: 16 list prices per million tokens for the same models, as reported by OpenRouter’s public API, with the gap stated.",
      "verdict": "DeepInfra and Parasail share 16 reported list prices from one study. No row names a winner: a reported list price has no interval, so each gap is stated, not ranked. Prices change often; the snapshot date is in each row’s context. The rows are 1 tie and 15 unclear; each row says why.",
      "rows": [
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Input)",
          "aValue": 0.037,
          "bValue": 0.1,
          "unit": "usd",
          "aDisplay": "$0.037",
          "bDisplay": "$0.10",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.037 vs $0.10, 2.7x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "bf16 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Output)",
          "aValue": 0.17,
          "bValue": 0.75,
          "unit": "usd",
          "aDisplay": "$0.17",
          "bDisplay": "$0.75",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.17 vs $0.75, 4.4x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "bf16 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Llama 3.3 70B Instruct: price per million tokens by provider (Input)",
          "aValue": 0.1,
          "bValue": 0.22,
          "unit": "usd",
          "aDisplay": "$0.10",
          "bDisplay": "$0.22",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.10 vs $0.22, 2.2x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-llama-3-3-70b-instruct",
          "aContext": "fp8 · Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Llama 3.3 70B Instruct: price per million tokens by provider (Output)",
          "aValue": 0.32,
          "bValue": 0.5,
          "unit": "usd",
          "aDisplay": "$0.32",
          "bDisplay": "$0.50",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.32 vs $0.50, 1.6x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-llama-3-3-70b-instruct",
          "aContext": "fp8 · Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "DeepSeek V4 Pro 0423: price per million tokens by provider (Input)",
          "aValue": 1.3,
          "bValue": 0.45,
          "unit": "usd",
          "aDisplay": "$1.30",
          "bDisplay": "$0.45",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($1.30 vs $0.45, 2.9x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-deepseek-v4-pro",
          "aContext": "fp8 · DeepSeek V4 Pro 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · DeepSeek V4 Pro 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "DeepSeek V4 Pro 0423: price per million tokens by provider (Output)",
          "aValue": 2.6,
          "bValue": 3.48,
          "unit": "usd",
          "aDisplay": "$2.60",
          "bDisplay": "$3.48",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($2.60 vs $3.48, 1.3x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-deepseek-v4-pro",
          "aContext": "fp8 · DeepSeek V4 Pro 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · DeepSeek V4 Pro 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "DeepSeek V4 Pro 0423: price per million tokens by provider (Cache read)",
          "aValue": 0.1,
          "bValue": 0.1,
          "unit": "usd",
          "aDisplay": "$0.10",
          "bDisplay": "$0.10",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-deepseek-v4-pro",
          "aContext": "fp8 · DeepSeek V4 Pro 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · DeepSeek V4 Pro 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "DeepSeek V4 Flash 0423: price per million tokens by provider (Input)",
          "aValue": 0.09,
          "bValue": 0.14,
          "unit": "usd",
          "aDisplay": "$0.090",
          "bDisplay": "$0.14",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.090 vs $0.14, 1.6x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-deepseek-v4-flash",
          "aContext": "fp8 · DeepSeek V4 Flash 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · DeepSeek V4 Flash 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "DeepSeek V4 Flash 0423: price per million tokens by provider (Output)",
          "aValue": 0.18,
          "bValue": 0.28,
          "unit": "usd",
          "aDisplay": "$0.18",
          "bDisplay": "$0.28",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.18 vs $0.28, 1.6x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-deepseek-v4-flash",
          "aContext": "fp8 · DeepSeek V4 Flash 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · DeepSeek V4 Flash 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "DeepSeek V4 Flash 0423: price per million tokens by provider (Cache read)",
          "aValue": 0.018,
          "bValue": 0.07,
          "unit": "usd",
          "aDisplay": "$0.018",
          "bDisplay": "$0.070",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.018 vs $0.070, 3.9x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-deepseek-v4-flash",
          "aContext": "fp8 · DeepSeek V4 Flash 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · DeepSeek V4 Flash 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Kimi K3: price per million tokens by provider (Input)",
          "aValue": 2.85,
          "bValue": 3,
          "unit": "usd",
          "aDisplay": "$2.85",
          "bDisplay": "$3.00",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($2.85 vs $3.00, 1.1x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-kimi-k3",
          "aContext": "mxfp4 · Kimi K3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp4 · Kimi K3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Kimi K3: price per million tokens by provider (Output)",
          "aValue": 14.25,
          "bValue": 15,
          "unit": "usd",
          "aDisplay": "$14.25",
          "bDisplay": "$15.00",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($14.25 vs $15.00, 1.1x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-kimi-k3",
          "aContext": "mxfp4 · Kimi K3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp4 · Kimi K3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Kimi K3: price per million tokens by provider (Cache read)",
          "aValue": 0.285,
          "bValue": 0.3,
          "unit": "usd",
          "aDisplay": "$0.28",
          "bDisplay": "$0.30",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.28 vs $0.30, 1.1x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-kimi-k3",
          "aContext": "mxfp4 · Kimi K3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp4 · Kimi K3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "GLM 5.3: price per million tokens by provider (Input)",
          "aValue": 0.5625,
          "bValue": 1.4,
          "unit": "usd",
          "aDisplay": "$0.56",
          "bDisplay": "$1.40",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.56 vs $1.40, 2.5x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "aContext": "fp4 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "GLM 5.3: price per million tokens by provider (Output)",
          "aValue": 2.5,
          "bValue": 4.4,
          "unit": "usd",
          "aDisplay": "$2.50",
          "bDisplay": "$4.40",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($2.50 vs $4.40, 1.8x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "aContext": "fp4 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "GLM 5.3: price per million tokens by provider (Cache read)",
          "aValue": 0.125,
          "bValue": 0.26,
          "unit": "usd",
          "aDisplay": "$0.13",
          "bDisplay": "$0.26",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.13 vs $0.26, 2.1x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "aContext": "fp4 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "amazon-bedrock-vs-claude-platform-on-aws",
      "a": "amazon-bedrock",
      "b": "claude-platform-on-aws",
      "title": "Amazon Bedrock vs Claude Platform on AWS",
      "seoTitle": "Amazon Bedrock vs Claude Platform on AWS: price per model",
      "description": "Amazon Bedrock vs Claude Platform on AWS: 15 list prices per million tokens for the same models, as reported by OpenRouter’s public API, with the gap stated.",
      "verdict": "Amazon Bedrock and Claude Platform on AWS share 15 reported list prices from one study. No row names a winner: a reported list price has no interval, so each gap is stated, not ranked. Prices change often; the snapshot date is in each row’s context. The rows are 15 ties; each row says why.",
      "rows": [
        {
          "metric": "Claude Sonnet 5: price per million tokens by provider (Input)",
          "aValue": 2,
          "bValue": 2,
          "unit": "usd",
          "aDisplay": "$2.00",
          "bDisplay": "$2.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-sonnet-5",
          "aContext": "Claude Sonnet 5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Sonnet 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Sonnet 5: price per million tokens by provider (Output)",
          "aValue": 10,
          "bValue": 10,
          "unit": "usd",
          "aDisplay": "$10.00",
          "bDisplay": "$10.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-sonnet-5",
          "aContext": "Claude Sonnet 5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Sonnet 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Sonnet 5: price per million tokens by provider (Cache read)",
          "aValue": 0.2,
          "bValue": 0.2,
          "unit": "usd",
          "aDisplay": "$0.20",
          "bDisplay": "$0.20",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-sonnet-5",
          "aContext": "Claude Sonnet 5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Sonnet 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Sonnet 5.5: price per million tokens by provider (Input)",
          "aValue": 2,
          "bValue": 2,
          "unit": "usd",
          "aDisplay": "$2.00",
          "bDisplay": "$2.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-sonnet-5-5",
          "aContext": "Claude Sonnet 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Sonnet 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Sonnet 5.5: price per million tokens by provider (Output)",
          "aValue": 10,
          "bValue": 10,
          "unit": "usd",
          "aDisplay": "$10.00",
          "bDisplay": "$10.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-sonnet-5-5",
          "aContext": "Claude Sonnet 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Sonnet 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Sonnet 5.5: price per million tokens by provider (Cache read)",
          "aValue": 0.2,
          "bValue": 0.2,
          "unit": "usd",
          "aDisplay": "$0.20",
          "bDisplay": "$0.20",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-sonnet-5-5",
          "aContext": "Claude Sonnet 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Sonnet 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Opus 4.8: price per million tokens by provider (Input)",
          "aValue": 5,
          "bValue": 5,
          "unit": "usd",
          "aDisplay": "$5.00",
          "bDisplay": "$5.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-4-8",
          "aContext": "Claude Opus 4.8 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Opus 4.8 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Opus 4.8: price per million tokens by provider (Output)",
          "aValue": 25,
          "bValue": 25,
          "unit": "usd",
          "aDisplay": "$25.00",
          "bDisplay": "$25.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-4-8",
          "aContext": "Claude Opus 4.8 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Opus 4.8 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Opus 4.8: price per million tokens by provider (Cache read)",
          "aValue": 0.5,
          "bValue": 0.5,
          "unit": "usd",
          "aDisplay": "$0.50",
          "bDisplay": "$0.50",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-4-8",
          "aContext": "Claude Opus 4.8 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Opus 4.8 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Opus 5: price per million tokens by provider (Input)",
          "aValue": 5,
          "bValue": 5,
          "unit": "usd",
          "aDisplay": "$5.00",
          "bDisplay": "$5.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-5",
          "aContext": "Claude Opus 5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Opus 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Opus 5: price per million tokens by provider (Output)",
          "aValue": 25,
          "bValue": 25,
          "unit": "usd",
          "aDisplay": "$25.00",
          "bDisplay": "$25.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-5",
          "aContext": "Claude Opus 5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Opus 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Opus 5: price per million tokens by provider (Cache read)",
          "aValue": 0.5,
          "bValue": 0.5,
          "unit": "usd",
          "aDisplay": "$0.50",
          "bDisplay": "$0.50",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-5",
          "aContext": "Claude Opus 5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Opus 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Opus 5.5: price per million tokens by provider (Input)",
          "aValue": 4,
          "bValue": 4,
          "unit": "usd",
          "aDisplay": "$4.00",
          "bDisplay": "$4.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-5-5",
          "aContext": "Claude Opus 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Opus 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Opus 5.5: price per million tokens by provider (Output)",
          "aValue": 20,
          "bValue": 20,
          "unit": "usd",
          "aDisplay": "$20.00",
          "bDisplay": "$20.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-5-5",
          "aContext": "Claude Opus 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Opus 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Opus 5.5: price per million tokens by provider (Cache read)",
          "aValue": 0.2,
          "bValue": 0.2,
          "unit": "usd",
          "aDisplay": "$0.20",
          "bDisplay": "$0.20",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-5-5",
          "aContext": "Claude Opus 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Opus 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "anthropic-vs-claude-platform-on-aws",
      "a": "anthropic",
      "b": "claude-platform-on-aws",
      "title": "Anthropic vs Claude Platform on AWS",
      "seoTitle": "Anthropic vs Claude Platform on AWS: price per model",
      "description": "Anthropic vs Claude Platform on AWS: 15 list prices per million tokens for the same models, as reported by OpenRouter’s public API, with the gap stated.",
      "verdict": "Anthropic and Claude Platform on AWS share 15 reported list prices from one study. No row names a winner: a reported list price has no interval, so each gap is stated, not ranked. Prices change often; the snapshot date is in each row’s context. The rows are 15 ties; each row says why.",
      "rows": [
        {
          "metric": "Claude Sonnet 5: price per million tokens by provider (Input)",
          "aValue": 2,
          "bValue": 2,
          "unit": "usd",
          "aDisplay": "$2.00",
          "bDisplay": "$2.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-sonnet-5",
          "aContext": "Claude Sonnet 5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Sonnet 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Sonnet 5: price per million tokens by provider (Output)",
          "aValue": 10,
          "bValue": 10,
          "unit": "usd",
          "aDisplay": "$10.00",
          "bDisplay": "$10.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-sonnet-5",
          "aContext": "Claude Sonnet 5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Sonnet 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Sonnet 5: price per million tokens by provider (Cache read)",
          "aValue": 0.2,
          "bValue": 0.2,
          "unit": "usd",
          "aDisplay": "$0.20",
          "bDisplay": "$0.20",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-sonnet-5",
          "aContext": "Claude Sonnet 5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Sonnet 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Sonnet 5.5: price per million tokens by provider (Input)",
          "aValue": 2,
          "bValue": 2,
          "unit": "usd",
          "aDisplay": "$2.00",
          "bDisplay": "$2.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-sonnet-5-5",
          "aContext": "Claude Sonnet 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Sonnet 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Sonnet 5.5: price per million tokens by provider (Output)",
          "aValue": 10,
          "bValue": 10,
          "unit": "usd",
          "aDisplay": "$10.00",
          "bDisplay": "$10.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-sonnet-5-5",
          "aContext": "Claude Sonnet 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Sonnet 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Sonnet 5.5: price per million tokens by provider (Cache read)",
          "aValue": 0.2,
          "bValue": 0.2,
          "unit": "usd",
          "aDisplay": "$0.20",
          "bDisplay": "$0.20",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-sonnet-5-5",
          "aContext": "Claude Sonnet 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Sonnet 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Opus 4.8: price per million tokens by provider (Input)",
          "aValue": 5,
          "bValue": 5,
          "unit": "usd",
          "aDisplay": "$5.00",
          "bDisplay": "$5.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-4-8",
          "aContext": "Claude Opus 4.8 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Opus 4.8 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Opus 4.8: price per million tokens by provider (Output)",
          "aValue": 25,
          "bValue": 25,
          "unit": "usd",
          "aDisplay": "$25.00",
          "bDisplay": "$25.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-4-8",
          "aContext": "Claude Opus 4.8 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Opus 4.8 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Opus 4.8: price per million tokens by provider (Cache read)",
          "aValue": 0.5,
          "bValue": 0.5,
          "unit": "usd",
          "aDisplay": "$0.50",
          "bDisplay": "$0.50",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-4-8",
          "aContext": "Claude Opus 4.8 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Opus 4.8 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Opus 5: price per million tokens by provider (Input)",
          "aValue": 5,
          "bValue": 5,
          "unit": "usd",
          "aDisplay": "$5.00",
          "bDisplay": "$5.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-5",
          "aContext": "Claude Opus 5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Opus 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Opus 5: price per million tokens by provider (Output)",
          "aValue": 25,
          "bValue": 25,
          "unit": "usd",
          "aDisplay": "$25.00",
          "bDisplay": "$25.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-5",
          "aContext": "Claude Opus 5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Opus 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Opus 5: price per million tokens by provider (Cache read)",
          "aValue": 0.5,
          "bValue": 0.5,
          "unit": "usd",
          "aDisplay": "$0.50",
          "bDisplay": "$0.50",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-5",
          "aContext": "Claude Opus 5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Opus 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Opus 5.5: price per million tokens by provider (Input)",
          "aValue": 4,
          "bValue": 4,
          "unit": "usd",
          "aDisplay": "$4.00",
          "bDisplay": "$4.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-5-5",
          "aContext": "Claude Opus 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Opus 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Opus 5.5: price per million tokens by provider (Output)",
          "aValue": 20,
          "bValue": 20,
          "unit": "usd",
          "aDisplay": "$20.00",
          "bDisplay": "$20.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-5-5",
          "aContext": "Claude Opus 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Opus 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Opus 5.5: price per million tokens by provider (Cache read)",
          "aValue": 0.2,
          "bValue": 0.2,
          "unit": "usd",
          "aDisplay": "$0.20",
          "bDisplay": "$0.20",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-5-5",
          "aContext": "Claude Opus 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Opus 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "azure-vs-claude-platform-on-aws",
      "a": "azure",
      "b": "claude-platform-on-aws",
      "title": "Azure vs Claude Platform on AWS",
      "seoTitle": "Azure vs Claude Platform on AWS: price per model",
      "description": "Azure vs Claude Platform on AWS: 15 list prices per million tokens for the same models, as reported by OpenRouter’s public API, with the gap stated.",
      "verdict": "Azure and Claude Platform on AWS share 15 reported list prices from one study. No row names a winner: a reported list price has no interval, so each gap is stated, not ranked. Prices change often; the snapshot date is in each row’s context. The rows are 15 ties; each row says why.",
      "rows": [
        {
          "metric": "Claude Sonnet 5: price per million tokens by provider (Input)",
          "aValue": 2,
          "bValue": 2,
          "unit": "usd",
          "aDisplay": "$2.00",
          "bDisplay": "$2.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-sonnet-5",
          "aContext": "Claude Sonnet 5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Sonnet 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Sonnet 5: price per million tokens by provider (Output)",
          "aValue": 10,
          "bValue": 10,
          "unit": "usd",
          "aDisplay": "$10.00",
          "bDisplay": "$10.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-sonnet-5",
          "aContext": "Claude Sonnet 5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Sonnet 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Sonnet 5: price per million tokens by provider (Cache read)",
          "aValue": 0.2,
          "bValue": 0.2,
          "unit": "usd",
          "aDisplay": "$0.20",
          "bDisplay": "$0.20",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-sonnet-5",
          "aContext": "Claude Sonnet 5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Sonnet 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Sonnet 5.5: price per million tokens by provider (Input)",
          "aValue": 2,
          "bValue": 2,
          "unit": "usd",
          "aDisplay": "$2.00",
          "bDisplay": "$2.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-sonnet-5-5",
          "aContext": "Claude Sonnet 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Sonnet 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Sonnet 5.5: price per million tokens by provider (Output)",
          "aValue": 10,
          "bValue": 10,
          "unit": "usd",
          "aDisplay": "$10.00",
          "bDisplay": "$10.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-sonnet-5-5",
          "aContext": "Claude Sonnet 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Sonnet 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Sonnet 5.5: price per million tokens by provider (Cache read)",
          "aValue": 0.2,
          "bValue": 0.2,
          "unit": "usd",
          "aDisplay": "$0.20",
          "bDisplay": "$0.20",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-sonnet-5-5",
          "aContext": "Claude Sonnet 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Sonnet 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Opus 4.8: price per million tokens by provider (Input)",
          "aValue": 5,
          "bValue": 5,
          "unit": "usd",
          "aDisplay": "$5.00",
          "bDisplay": "$5.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-4-8",
          "aContext": "Claude Opus 4.8 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Opus 4.8 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Opus 4.8: price per million tokens by provider (Output)",
          "aValue": 25,
          "bValue": 25,
          "unit": "usd",
          "aDisplay": "$25.00",
          "bDisplay": "$25.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-4-8",
          "aContext": "Claude Opus 4.8 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Opus 4.8 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Opus 4.8: price per million tokens by provider (Cache read)",
          "aValue": 0.5,
          "bValue": 0.5,
          "unit": "usd",
          "aDisplay": "$0.50",
          "bDisplay": "$0.50",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-4-8",
          "aContext": "Claude Opus 4.8 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Opus 4.8 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Opus 5: price per million tokens by provider (Input)",
          "aValue": 5,
          "bValue": 5,
          "unit": "usd",
          "aDisplay": "$5.00",
          "bDisplay": "$5.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-5",
          "aContext": "Claude Opus 5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Opus 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Opus 5: price per million tokens by provider (Output)",
          "aValue": 25,
          "bValue": 25,
          "unit": "usd",
          "aDisplay": "$25.00",
          "bDisplay": "$25.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-5",
          "aContext": "Claude Opus 5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Opus 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Opus 5: price per million tokens by provider (Cache read)",
          "aValue": 0.5,
          "bValue": 0.5,
          "unit": "usd",
          "aDisplay": "$0.50",
          "bDisplay": "$0.50",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-5",
          "aContext": "Claude Opus 5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Opus 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Opus 5.5: price per million tokens by provider (Input)",
          "aValue": 4,
          "bValue": 4,
          "unit": "usd",
          "aDisplay": "$4.00",
          "bDisplay": "$4.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-5-5",
          "aContext": "Claude Opus 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Opus 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Opus 5.5: price per million tokens by provider (Output)",
          "aValue": 20,
          "bValue": 20,
          "unit": "usd",
          "aDisplay": "$20.00",
          "bDisplay": "$20.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-5-5",
          "aContext": "Claude Opus 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Opus 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Opus 5.5: price per million tokens by provider (Cache read)",
          "aValue": 0.2,
          "bValue": 0.2,
          "unit": "usd",
          "aDisplay": "$0.20",
          "bDisplay": "$0.20",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-5-5",
          "aContext": "Claude Opus 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Opus 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "google-vertex-vs-claude-platform-on-aws",
      "a": "google-vertex",
      "b": "claude-platform-on-aws",
      "title": "Google Vertex AI vs Claude Platform on AWS",
      "seoTitle": "Google Vertex AI vs Claude Platform on AWS: price per model",
      "description": "Google Vertex AI vs Claude Platform on AWS: 15 list prices per million tokens for the same models, as reported by OpenRouter’s public API, with the gap stated.",
      "verdict": "Google Vertex AI and Claude Platform on AWS share 15 reported list prices from one study. No row names a winner: a reported list price has no interval, so each gap is stated, not ranked. Prices change often; the snapshot date is in each row’s context. The rows are 15 ties; each row says why.",
      "rows": [
        {
          "metric": "Claude Sonnet 5: price per million tokens by provider (Input)",
          "aValue": 2,
          "bValue": 2,
          "unit": "usd",
          "aDisplay": "$2.00",
          "bDisplay": "$2.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-sonnet-5",
          "aContext": "Claude Sonnet 5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Sonnet 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Sonnet 5: price per million tokens by provider (Output)",
          "aValue": 10,
          "bValue": 10,
          "unit": "usd",
          "aDisplay": "$10.00",
          "bDisplay": "$10.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-sonnet-5",
          "aContext": "Claude Sonnet 5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Sonnet 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Sonnet 5: price per million tokens by provider (Cache read)",
          "aValue": 0.2,
          "bValue": 0.2,
          "unit": "usd",
          "aDisplay": "$0.20",
          "bDisplay": "$0.20",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-sonnet-5",
          "aContext": "Claude Sonnet 5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Sonnet 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Sonnet 5.5: price per million tokens by provider (Input)",
          "aValue": 2,
          "bValue": 2,
          "unit": "usd",
          "aDisplay": "$2.00",
          "bDisplay": "$2.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-sonnet-5-5",
          "aContext": "Claude Sonnet 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Sonnet 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Sonnet 5.5: price per million tokens by provider (Output)",
          "aValue": 10,
          "bValue": 10,
          "unit": "usd",
          "aDisplay": "$10.00",
          "bDisplay": "$10.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-sonnet-5-5",
          "aContext": "Claude Sonnet 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Sonnet 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Sonnet 5.5: price per million tokens by provider (Cache read)",
          "aValue": 0.2,
          "bValue": 0.2,
          "unit": "usd",
          "aDisplay": "$0.20",
          "bDisplay": "$0.20",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-sonnet-5-5",
          "aContext": "Claude Sonnet 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Sonnet 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Opus 4.8: price per million tokens by provider (Input)",
          "aValue": 5,
          "bValue": 5,
          "unit": "usd",
          "aDisplay": "$5.00",
          "bDisplay": "$5.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-4-8",
          "aContext": "Claude Opus 4.8 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Opus 4.8 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Opus 4.8: price per million tokens by provider (Output)",
          "aValue": 25,
          "bValue": 25,
          "unit": "usd",
          "aDisplay": "$25.00",
          "bDisplay": "$25.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-4-8",
          "aContext": "Claude Opus 4.8 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Opus 4.8 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Opus 4.8: price per million tokens by provider (Cache read)",
          "aValue": 0.5,
          "bValue": 0.5,
          "unit": "usd",
          "aDisplay": "$0.50",
          "bDisplay": "$0.50",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-4-8",
          "aContext": "Claude Opus 4.8 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Opus 4.8 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Opus 5: price per million tokens by provider (Input)",
          "aValue": 5,
          "bValue": 5,
          "unit": "usd",
          "aDisplay": "$5.00",
          "bDisplay": "$5.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-5",
          "aContext": "Claude Opus 5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Opus 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Opus 5: price per million tokens by provider (Output)",
          "aValue": 25,
          "bValue": 25,
          "unit": "usd",
          "aDisplay": "$25.00",
          "bDisplay": "$25.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-5",
          "aContext": "Claude Opus 5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Opus 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Opus 5: price per million tokens by provider (Cache read)",
          "aValue": 0.5,
          "bValue": 0.5,
          "unit": "usd",
          "aDisplay": "$0.50",
          "bDisplay": "$0.50",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-5",
          "aContext": "Claude Opus 5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Opus 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Opus 5.5: price per million tokens by provider (Input)",
          "aValue": 4,
          "bValue": 4,
          "unit": "usd",
          "aDisplay": "$4.00",
          "bDisplay": "$4.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-5-5",
          "aContext": "Claude Opus 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Opus 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Opus 5.5: price per million tokens by provider (Output)",
          "aValue": 20,
          "bValue": 20,
          "unit": "usd",
          "aDisplay": "$20.00",
          "bDisplay": "$20.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-5-5",
          "aContext": "Claude Opus 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Opus 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Claude Opus 5.5: price per million tokens by provider (Cache read)",
          "aValue": 0.2,
          "bValue": 0.2,
          "unit": "usd",
          "aDisplay": "$0.20",
          "bDisplay": "$0.20",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-claude-opus-5-5",
          "aContext": "Claude Opus 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Claude Opus 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "parasail-vs-novita",
      "a": "parasail",
      "b": "novita",
      "title": "Parasail vs Novita AI",
      "seoTitle": "Parasail vs Novita AI: price per model",
      "description": "Parasail vs Novita AI: 15 list prices per million tokens for the same models, as reported by OpenRouter’s public API, with the gap stated.",
      "verdict": "Parasail and Novita AI share 15 reported list prices from one study. No row names a winner: a reported list price has no interval, so each gap is stated, not ranked. Prices change often; the snapshot date is in each row’s context. The rows are 2 ties and 13 unclear; each row says why.",
      "rows": [
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Input)",
          "aValue": 0.1,
          "bValue": 0.05,
          "unit": "usd",
          "aDisplay": "$0.10",
          "bDisplay": "$0.050",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.10 vs $0.050, 2.0x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Output)",
          "aValue": 0.75,
          "bValue": 0.25,
          "unit": "usd",
          "aDisplay": "$0.75",
          "bDisplay": "$0.25",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.75 vs $0.25, 3.0x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Llama 4 Maverick: price per million tokens by provider (Input)",
          "aValue": 0.35,
          "bValue": 0.27,
          "unit": "usd",
          "aDisplay": "$0.35",
          "bDisplay": "$0.27",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.35 vs $0.27, 1.3x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-llama-4-maverick",
          "aContext": "fp8 · Llama 4 Maverick · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · Llama 4 Maverick · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Llama 4 Maverick: price per million tokens by provider (Output)",
          "aValue": 1,
          "bValue": 0.85,
          "unit": "usd",
          "aDisplay": "$1.00",
          "bDisplay": "$0.85",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($1.00 vs $0.85, 1.2x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-llama-4-maverick",
          "aContext": "fp8 · Llama 4 Maverick · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · Llama 4 Maverick · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Llama 3.3 70B Instruct: price per million tokens by provider (Input)",
          "aValue": 0.22,
          "bValue": 0.135,
          "unit": "usd",
          "aDisplay": "$0.22",
          "bDisplay": "$0.14",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.22 vs $0.14, 1.6x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-llama-3-3-70b-instruct",
          "aContext": "fp8 · Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "bf16 · Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Llama 3.3 70B Instruct: price per million tokens by provider (Output)",
          "aValue": 0.5,
          "bValue": 0.4,
          "unit": "usd",
          "aDisplay": "$0.50",
          "bDisplay": "$0.40",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.50 vs $0.40, 1.3x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-llama-3-3-70b-instruct",
          "aContext": "fp8 · Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "bf16 · Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "DeepSeek V4 Pro 0423: price per million tokens by provider (Input)",
          "aValue": 0.45,
          "bValue": 1.6,
          "unit": "usd",
          "aDisplay": "$0.45",
          "bDisplay": "$1.60",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.45 vs $1.60, 3.6x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-deepseek-v4-pro",
          "aContext": "fp8 · DeepSeek V4 Pro 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · DeepSeek V4 Pro 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "DeepSeek V4 Pro 0423: price per million tokens by provider (Output)",
          "aValue": 3.48,
          "bValue": 3.2,
          "unit": "usd",
          "aDisplay": "$3.48",
          "bDisplay": "$3.20",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($3.48 vs $3.20, 1.1x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-deepseek-v4-pro",
          "aContext": "fp8 · DeepSeek V4 Pro 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · DeepSeek V4 Pro 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "DeepSeek V4 Pro 0423: price per million tokens by provider (Cache read)",
          "aValue": 0.1,
          "bValue": 0.135,
          "unit": "usd",
          "aDisplay": "$0.10",
          "bDisplay": "$0.14",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.10 vs $0.14, 1.4x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-deepseek-v4-pro",
          "aContext": "fp8 · DeepSeek V4 Pro 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · DeepSeek V4 Pro 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "DeepSeek V4 Flash 0423: price per million tokens by provider (Input)",
          "aValue": 0.14,
          "bValue": 0.14,
          "unit": "usd",
          "aDisplay": "$0.14",
          "bDisplay": "$0.14",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-deepseek-v4-flash",
          "aContext": "fp8 · DeepSeek V4 Flash 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · DeepSeek V4 Flash 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "DeepSeek V4 Flash 0423: price per million tokens by provider (Output)",
          "aValue": 0.28,
          "bValue": 0.28,
          "unit": "usd",
          "aDisplay": "$0.28",
          "bDisplay": "$0.28",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-deepseek-v4-flash",
          "aContext": "fp8 · DeepSeek V4 Flash 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · DeepSeek V4 Flash 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "DeepSeek V4 Flash 0423: price per million tokens by provider (Cache read)",
          "aValue": 0.07,
          "bValue": 0.028,
          "unit": "usd",
          "aDisplay": "$0.070",
          "bDisplay": "$0.028",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.070 vs $0.028, 2.5x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-deepseek-v4-flash",
          "aContext": "fp8 · DeepSeek V4 Flash 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · DeepSeek V4 Flash 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "GLM 5.3: price per million tokens by provider (Input)",
          "aValue": 1.4,
          "bValue": 0.42,
          "unit": "usd",
          "aDisplay": "$1.40",
          "bDisplay": "$0.42",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($1.40 vs $0.42, 3.3x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "aContext": "fp8 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "GLM 5.3: price per million tokens by provider (Output)",
          "aValue": 4.4,
          "bValue": 1.32,
          "unit": "usd",
          "aDisplay": "$4.40",
          "bDisplay": "$1.32",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($4.40 vs $1.32, 3.3x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "aContext": "fp8 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "GLM 5.3: price per million tokens by provider (Cache read)",
          "aValue": 0.26,
          "bValue": 0.078,
          "unit": "usd",
          "aDisplay": "$0.26",
          "bDisplay": "$0.078",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.26 vs $0.078, 3.3x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "aContext": "fp8 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "claude-haiku-4-5-vs-gpt-6-luna-codex-cli",
      "a": "claude-haiku-4-5",
      "b": "gpt-6-luna-codex-cli",
      "title": "Claude Haiku 4.5 vs GPT-6 Luna (Codex CLI)",
      "seoTitle": "Claude Haiku 4.5 vs GPT-6 Luna (Codex CLI): benchmarks",
      "description": "Claude Haiku 4.5 vs GPT-6 Luna (Codex CLI): 14 measured metrics from 2 studies, with sample sizes, intervals and every failure counted.",
      "verdict": "Claude Haiku 4.5 and GPT-6 Luna (Codex CLI) share 14 measured metrics and 3 list-price calculations from 2 studies. GPT-6 Luna (Codex CLI) leads on 1 row: Total time per attempt: single call vs agent loop, 5.16 s vs 39.0 s. On those rows the run ranges do not overlap; only a 95% interval is a confidence interval. The other rows are 9 ties and 7 unclear; each row says why. Calculation rows are derived from list prices and recorded counts; they are not bills or runs. Every row ran the two sides through different routes (for example Claude Code vs Codex CLI), so they compare route + model pairs, not models alone; the contexts name the route. Some rows rest on small samples (n = 2 at the smallest).",
      "rows": [
        {
          "metric": "Strict pass rate: single call vs agent loop on eight hard tasks",
          "aValue": 0.4583,
          "bValue": 0.625,
          "unit": "rate",
          "aDisplay": "46% (11/24)",
          "bDisplay": "63% (10/16)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Haiku 4.5 28% to 65%; GPT-6 Luna (Codex CLI) 39% to 82%), so this sample cannot separate them.",
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-pass-rate",
          "aN": 24,
          "bN": 16,
          "aContext": "Claude Code · single call",
          "bContext": "Codex CLI · single call",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.2789,
            0.6493
          ],
          "bRange": [
            0.3864,
            0.8152
          ]
        },
        {
          "metric": "Strict passes per task: single call vs agent loop: Interval merge fix",
          "aValue": 1,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "100% (3/3)",
          "bDisplay": "100% (2/2)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Haiku 4.5 44% to 100%; GPT-6 Luna (Codex CLI) 34% to 100%), so this sample cannot separate them.",
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "aN": 3,
          "bN": 2,
          "aContext": "Claude Code · single call",
          "bContext": "Codex CLI · single call",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.4385,
            1
          ],
          "bRange": [
            0.3424,
            1
          ]
        },
        {
          "metric": "Strict passes per task: single call vs agent loop: DST day-length fix",
          "aValue": 0.3333,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "33% (1/3)",
          "bDisplay": "100% (2/2)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Haiku 4.5 6% to 79%; GPT-6 Luna (Codex CLI) 34% to 100%), so this sample cannot separate them.",
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "aN": 3,
          "bN": 2,
          "aContext": "Claude Code · single call",
          "bContext": "Codex CLI · single call",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.0615,
            0.7923
          ],
          "bRange": [
            0.3424,
            1
          ]
        },
        {
          "metric": "Strict passes per task: single call vs agent loop: CSV parser",
          "aValue": 0.6667,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "67% (2/3)",
          "bDisplay": "100% (2/2)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Haiku 4.5 21% to 94%; GPT-6 Luna (Codex CLI) 34% to 100%), so this sample cannot separate them.",
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "aN": 3,
          "bN": 2,
          "aContext": "Claude Code · single call",
          "bContext": "Codex CLI · single call",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.2077,
            0.9385
          ],
          "bRange": [
            0.3424,
            1
          ]
        },
        {
          "metric": "Strict passes per task: single call vs agent loop: Event-loop order",
          "aValue": 0,
          "bValue": 0,
          "unit": "rate",
          "aDisplay": "0% (0/3)",
          "bDisplay": "0% (0/2)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Haiku 4.5 0% to 56%; GPT-6 Luna (Codex CLI) 0% to 66%), so this sample cannot separate them.",
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "aN": 3,
          "bN": 2,
          "aContext": "Claude Code · single call",
          "bContext": "Codex CLI · single call",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0,
            0.5615
          ],
          "bRange": [
            0,
            0.6576
          ]
        },
        {
          "metric": "Strict passes per task: single call vs agent loop: Room schedule",
          "aValue": 0,
          "bValue": 0.5,
          "unit": "rate",
          "aDisplay": "0% (0/3)",
          "bDisplay": "50% (1/2)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Haiku 4.5 0% to 56%; GPT-6 Luna (Codex CLI) 9% to 91%), so this sample cannot separate them.",
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "aN": 3,
          "bN": 2,
          "aContext": "Claude Code · single call",
          "bContext": "Codex CLI · single call",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0,
            0.5615
          ],
          "bRange": [
            0.0945,
            0.9055
          ]
        },
        {
          "metric": "Strict passes per task: single call vs agent loop: SemVer regex",
          "aValue": 1,
          "bValue": 0.5,
          "unit": "rate",
          "aDisplay": "100% (3/3)",
          "bDisplay": "50% (1/2)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Haiku 4.5 44% to 100%; GPT-6 Luna (Codex CLI) 9% to 91%), so this sample cannot separate them.",
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "aN": 3,
          "bN": 2,
          "aContext": "Claude Code · single call",
          "bContext": "Codex CLI · single call",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.4385,
            1
          ],
          "bRange": [
            0.0945,
            0.9055
          ]
        },
        {
          "metric": "Strict passes per task: single call vs agent loop: Money refactor",
          "aValue": 0.6667,
          "bValue": 0,
          "unit": "rate",
          "aDisplay": "67% (2/3)",
          "bDisplay": "0% (0/2)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Haiku 4.5 21% to 94%; GPT-6 Luna (Codex CLI) 0% to 66%), so this sample cannot separate them.",
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "aN": 3,
          "bN": 2,
          "aContext": "Claude Code · single call",
          "bContext": "Codex CLI · single call",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.2077,
            0.9385
          ],
          "bRange": [
            0,
            0.6576
          ]
        },
        {
          "metric": "Strict passes per task: single call vs agent loop: SQL report",
          "aValue": 0,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "0% (0/3)",
          "bDisplay": "100% (2/2)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Haiku 4.5 0% to 56%; GPT-6 Luna (Codex CLI) 34% to 100%), so this sample cannot separate them.",
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "aN": 3,
          "bN": 2,
          "aContext": "Claude Code · single call",
          "bContext": "Codex CLI · single call",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0,
            0.5615
          ],
          "bRange": [
            0.3424,
            1
          ]
        },
        {
          "metric": "Total time per attempt: single call vs agent loop",
          "aValue": 39.01,
          "bValue": 5.16,
          "unit": "seconds",
          "aDisplay": "39.0 s",
          "bDisplay": "5.16 s",
          "winner": "b",
          "basis": "The run ranges (fastest to slowest) do not overlap (Claude Haiku 4.5 15.3 s to 75.1 s; GPT-6 Luna (Codex CLI) 3.59 s to 11.3 s). A range is not a confidence interval.",
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-total-time",
          "aN": 24,
          "bN": 16,
          "aContext": "Claude Code · single call",
          "bContext": "Codex CLI · single call",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            15.27,
            75.13
          ],
          "bRange": [
            3.59,
            11.32
          ]
        },
        {
          "metric": "Tokens per attempt: single call vs agent loop (Input tokens (cache reads included))",
          "aValue": 3941,
          "bValue": 11582,
          "unit": "tokens",
          "aDisplay": "3,941",
          "bDisplay": "11,582",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-tokens",
          "aN": 24,
          "bN": 16,
          "aContext": "Claude Code · single call",
          "bContext": "Codex CLI · single call",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            3879,
            4221
          ],
          "bRange": [
            11526,
            11818
          ]
        },
        {
          "metric": "Tokens per attempt: single call vs agent loop (Output tokens)",
          "aValue": 5064,
          "bValue": 345,
          "unit": "tokens",
          "aDisplay": "5,064",
          "bDisplay": "345",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-tokens",
          "aN": 24,
          "bN": 16,
          "aContext": "Claude Code · single call",
          "bContext": "Codex CLI · single call",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            1899,
            9321
          ],
          "bRange": [
            36,
            634
          ]
        },
        {
          "metric": "Tool calls per agent-loop attempt",
          "aValue": 3,
          "bValue": 0,
          "unit": "count",
          "aDisplay": "3",
          "bDisplay": "0",
          "winner": "unclear",
          "basis": "More or fewer count is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-tool-calls",
          "aN": 24,
          "bN": 14,
          "aContext": "Claude Code · agent loop",
          "bContext": "Codex CLI · agent loop",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            2,
            18
          ],
          "bRange": [
            0,
            1
          ]
        },
        {
          "metric": "List-price cost per strict pass: single call vs agent loop (calculation)",
          "aValue": 0.0672,
          "bValue": 0.00116,
          "unit": "usd",
          "aDisplay": "$0.067",
          "bDisplay": "$0.0012",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.067 vs $0.0012, 58x) is not tested against run-to-run variation.",
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-cost-per-pass",
          "aN": 24,
          "bN": 16,
          "aContext": "Claude Code · single call",
          "bContext": "Codex CLI · single call",
          "calculation": true
        },
        {
          "metric": "Time to first text: a 250-line answer, six models",
          "aValue": 4,
          "bValue": 3.3,
          "unit": "seconds",
          "aDisplay": "4.00 s",
          "bDisplay": "3.30 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Haiku 4.5 2.84 s to 6.38 s; GPT-6 Luna (Codex CLI) 3.19 s to 3.47 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "llm-speed-anatomy",
          "n": 4,
          "chartId": "speed-anatomy-first-text",
          "aN": 4,
          "bN": 4,
          "aContext": "Claude Code",
          "bContext": "Codex CLI · effort low",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            2.84,
            6.38
          ],
          "bRange": [
            3.19,
            3.47
          ]
        },
        {
          "metric": "Output speed after the first text: visible tokens per second (calculation)",
          "aValue": 153.2,
          "bValue": 129.1,
          "unit": "tokens",
          "aDisplay": "153",
          "bDisplay": "129",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "llm-speed-anatomy",
          "n": 4,
          "chartId": "speed-anatomy-output-speed",
          "aN": 4,
          "bN": 4,
          "aContext": "Claude Code",
          "bContext": "Codex CLI · effort low",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            152.6,
            216.1
          ],
          "bRange": [
            55.5,
            259.1
          ],
          "calculation": true
        },
        {
          "metric": "Output speed in characters per second after the first text (calculation)",
          "aValue": 547,
          "bValue": 524,
          "unit": "count",
          "aDisplay": "547",
          "bDisplay": "524",
          "winner": "unclear",
          "basis": "More or fewer count is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "llm-speed-anatomy",
          "chartId": "speed-anatomy-chars-per-second",
          "aN": 3,
          "bN": 4,
          "aContext": "Claude Code",
          "bContext": "Codex CLI · effort low",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            546,
            548
          ],
          "bRange": [
            225,
            1052
          ],
          "calculation": true
        }
      ]
    },
    {
      "slug": "claude-sonnet-5-5-vs-gpt-6-luna-codex-cli",
      "a": "claude-sonnet-5-5",
      "b": "gpt-6-luna-codex-cli",
      "title": "Claude Sonnet 5.5 vs GPT-6 Luna (Codex CLI)",
      "seoTitle": "Claude Sonnet 5.5 vs GPT-6 Luna (Codex CLI): benchmarks",
      "description": "Claude Sonnet 5.5 vs GPT-6 Luna (Codex CLI): 14 measured metrics from 2 studies, with sample sizes, intervals and every failure counted.",
      "verdict": "Claude Sonnet 5.5 and GPT-6 Luna (Codex CLI) share 14 measured metrics and 3 list-price calculations from 2 studies. Claude Sonnet 5.5 leads on 1 row: Strict pass rate: single call vs agent loop on eight hard tasks, 100% (24/24) vs 63% (10/16). On those rows the 95% intervals do not overlap. The other rows are 9 ties and 7 unclear; each row says why. Calculation rows are derived from list prices and recorded counts; they are not bills or runs. Every row ran the two sides through different routes (for example Claude Code vs Codex CLI), so they compare route + model pairs, not models alone; the contexts name the route. Some rows rest on small samples (n = 2 at the smallest).",
      "rows": [
        {
          "metric": "Strict pass rate: single call vs agent loop on eight hard tasks",
          "aValue": 1,
          "bValue": 0.625,
          "unit": "rate",
          "aDisplay": "100% (24/24)",
          "bDisplay": "63% (10/16)",
          "winner": "a",
          "basis": "The 95% intervals do not overlap (Claude Sonnet 5.5 86% to 100%; GPT-6 Luna (Codex CLI) 39% to 82%).",
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-pass-rate",
          "aN": 24,
          "bN": 16,
          "aContext": "Claude Code · single call",
          "bContext": "Codex CLI · single call",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.862,
            1
          ],
          "bRange": [
            0.3864,
            0.8152
          ]
        },
        {
          "metric": "Strict passes per task: single call vs agent loop: Interval merge fix",
          "aValue": 1,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "100% (3/3)",
          "bDisplay": "100% (2/2)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Sonnet 5.5 44% to 100%; GPT-6 Luna (Codex CLI) 34% to 100%), so this sample cannot separate them.",
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "aN": 3,
          "bN": 2,
          "aContext": "Claude Code · single call",
          "bContext": "Codex CLI · single call",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.4385,
            1
          ],
          "bRange": [
            0.3424,
            1
          ]
        },
        {
          "metric": "Strict passes per task: single call vs agent loop: DST day-length fix",
          "aValue": 1,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "100% (3/3)",
          "bDisplay": "100% (2/2)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Sonnet 5.5 44% to 100%; GPT-6 Luna (Codex CLI) 34% to 100%), so this sample cannot separate them.",
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "aN": 3,
          "bN": 2,
          "aContext": "Claude Code · single call",
          "bContext": "Codex CLI · single call",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.4385,
            1
          ],
          "bRange": [
            0.3424,
            1
          ]
        },
        {
          "metric": "Strict passes per task: single call vs agent loop: CSV parser",
          "aValue": 1,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "100% (3/3)",
          "bDisplay": "100% (2/2)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Sonnet 5.5 44% to 100%; GPT-6 Luna (Codex CLI) 34% to 100%), so this sample cannot separate them.",
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "aN": 3,
          "bN": 2,
          "aContext": "Claude Code · single call",
          "bContext": "Codex CLI · single call",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.4385,
            1
          ],
          "bRange": [
            0.3424,
            1
          ]
        },
        {
          "metric": "Strict passes per task: single call vs agent loop: Event-loop order",
          "aValue": 1,
          "bValue": 0,
          "unit": "rate",
          "aDisplay": "100% (3/3)",
          "bDisplay": "0% (0/2)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Sonnet 5.5 44% to 100%; GPT-6 Luna (Codex CLI) 0% to 66%), so this sample cannot separate them.",
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "aN": 3,
          "bN": 2,
          "aContext": "Claude Code · single call",
          "bContext": "Codex CLI · single call",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.4385,
            1
          ],
          "bRange": [
            0,
            0.6576
          ]
        },
        {
          "metric": "Strict passes per task: single call vs agent loop: Room schedule",
          "aValue": 1,
          "bValue": 0.5,
          "unit": "rate",
          "aDisplay": "100% (3/3)",
          "bDisplay": "50% (1/2)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Sonnet 5.5 44% to 100%; GPT-6 Luna (Codex CLI) 9% to 91%), so this sample cannot separate them.",
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "aN": 3,
          "bN": 2,
          "aContext": "Claude Code · single call",
          "bContext": "Codex CLI · single call",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.4385,
            1
          ],
          "bRange": [
            0.0945,
            0.9055
          ]
        },
        {
          "metric": "Strict passes per task: single call vs agent loop: SemVer regex",
          "aValue": 1,
          "bValue": 0.5,
          "unit": "rate",
          "aDisplay": "100% (3/3)",
          "bDisplay": "50% (1/2)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Sonnet 5.5 44% to 100%; GPT-6 Luna (Codex CLI) 9% to 91%), so this sample cannot separate them.",
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "aN": 3,
          "bN": 2,
          "aContext": "Claude Code · single call",
          "bContext": "Codex CLI · single call",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.4385,
            1
          ],
          "bRange": [
            0.0945,
            0.9055
          ]
        },
        {
          "metric": "Strict passes per task: single call vs agent loop: Money refactor",
          "aValue": 1,
          "bValue": 0,
          "unit": "rate",
          "aDisplay": "100% (3/3)",
          "bDisplay": "0% (0/2)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Sonnet 5.5 44% to 100%; GPT-6 Luna (Codex CLI) 0% to 66%), so this sample cannot separate them.",
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "aN": 3,
          "bN": 2,
          "aContext": "Claude Code · single call",
          "bContext": "Codex CLI · single call",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.4385,
            1
          ],
          "bRange": [
            0,
            0.6576
          ]
        },
        {
          "metric": "Strict passes per task: single call vs agent loop: SQL report",
          "aValue": 1,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "100% (3/3)",
          "bDisplay": "100% (2/2)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Sonnet 5.5 44% to 100%; GPT-6 Luna (Codex CLI) 34% to 100%), so this sample cannot separate them.",
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-by-task",
          "aN": 3,
          "bN": 2,
          "aContext": "Claude Code · single call",
          "bContext": "Codex CLI · single call",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.4385,
            1
          ],
          "bRange": [
            0.3424,
            1
          ]
        },
        {
          "metric": "Total time per attempt: single call vs agent loop",
          "aValue": 7.75,
          "bValue": 5.16,
          "unit": "seconds",
          "aDisplay": "7.75 s",
          "bDisplay": "5.16 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Sonnet 5.5 2.26 s to 34.8 s; GPT-6 Luna (Codex CLI) 3.59 s to 11.3 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-total-time",
          "aN": 24,
          "bN": 16,
          "aContext": "Claude Code · single call",
          "bContext": "Codex CLI · single call",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            2.26,
            34.79
          ],
          "bRange": [
            3.59,
            11.32
          ]
        },
        {
          "metric": "Tokens per attempt: single call vs agent loop (Input tokens (cache reads included))",
          "aValue": 2281,
          "bValue": 11582,
          "unit": "tokens",
          "aDisplay": "2,281",
          "bDisplay": "11,582",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-tokens",
          "aN": 24,
          "bN": 16,
          "aContext": "Claude Code · single call",
          "bContext": "Codex CLI · single call",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            2234,
            2669
          ],
          "bRange": [
            11526,
            11818
          ]
        },
        {
          "metric": "Tokens per attempt: single call vs agent loop (Output tokens)",
          "aValue": 1050,
          "bValue": 345,
          "unit": "tokens",
          "aDisplay": "1,050",
          "bDisplay": "345",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-tokens",
          "aN": 24,
          "bN": 16,
          "aContext": "Claude Code · single call",
          "bContext": "Codex CLI · single call",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            176,
            3895
          ],
          "bRange": [
            36,
            634
          ]
        },
        {
          "metric": "Tool calls per agent-loop attempt",
          "aValue": 0,
          "bValue": 0,
          "unit": "count",
          "aDisplay": "0",
          "bDisplay": "0",
          "winner": "tie",
          "basis": "Same value. More or fewer is not better by itself for this metric.",
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-tool-calls",
          "aN": 16,
          "bN": 14,
          "aContext": "Claude Code · agent loop",
          "bContext": "Codex CLI · agent loop",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            0,
            3
          ],
          "bRange": [
            0,
            1
          ]
        },
        {
          "metric": "List-price cost per strict pass: single call vs agent loop (calculation)",
          "aValue": 0.01435,
          "bValue": 0.00116,
          "unit": "usd",
          "aDisplay": "$0.014",
          "bDisplay": "$0.0012",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.014 vs $0.0012, 12x) is not tested against run-to-run variation.",
          "studySlug": "single-call-vs-agent-loop",
          "chartId": "agent-loop-cost-per-pass",
          "aN": 24,
          "bN": 16,
          "aContext": "Claude Code · single call",
          "bContext": "Codex CLI · single call",
          "calculation": true
        },
        {
          "metric": "Time to first text: a 250-line answer, six models",
          "aValue": 1.96,
          "bValue": 3.3,
          "unit": "seconds",
          "aDisplay": "1.96 s",
          "bDisplay": "3.30 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Sonnet 5.5 0.88 s to 4.09 s; GPT-6 Luna (Codex CLI) 3.19 s to 3.47 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "llm-speed-anatomy",
          "n": 4,
          "chartId": "speed-anatomy-first-text",
          "aN": 4,
          "bN": 4,
          "aContext": "Claude Code",
          "bContext": "Codex CLI · effort low",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            0.88,
            4.09
          ],
          "bRange": [
            3.19,
            3.47
          ]
        },
        {
          "metric": "Output speed after the first text: visible tokens per second (calculation)",
          "aValue": 231.7,
          "bValue": 129.1,
          "unit": "tokens",
          "aDisplay": "232",
          "bDisplay": "129",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "llm-speed-anatomy",
          "n": 4,
          "chartId": "speed-anatomy-output-speed",
          "aN": 4,
          "bN": 4,
          "aContext": "Claude Code",
          "bContext": "Codex CLI · effort low",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            230.3,
            233
          ],
          "bRange": [
            55.5,
            259.1
          ],
          "calculation": true
        },
        {
          "metric": "Output speed in characters per second after the first text (calculation)",
          "aValue": 517,
          "bValue": 524,
          "unit": "count",
          "aDisplay": "517",
          "bDisplay": "524",
          "winner": "unclear",
          "basis": "More or fewer count is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "llm-speed-anatomy",
          "n": 4,
          "chartId": "speed-anatomy-chars-per-second",
          "aN": 4,
          "bN": 4,
          "aContext": "Claude Code",
          "bContext": "Codex CLI · effort low",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            513,
            519
          ],
          "bRange": [
            225,
            1052
          ],
          "calculation": true
        }
      ]
    },
    {
      "slug": "deepinfra-vs-novita",
      "a": "deepinfra",
      "b": "novita",
      "title": "DeepInfra vs Novita AI",
      "seoTitle": "DeepInfra vs Novita AI: price per model",
      "description": "DeepInfra vs Novita AI: 13 list prices per million tokens for the same models, as reported by OpenRouter’s public API, with the gap stated.",
      "verdict": "DeepInfra and Novita AI share 13 reported list prices from one study. No row names a winner: a reported list price has no interval, so each gap is stated, not ranked. Prices change often; the snapshot date is in each row’s context. The rows are 13 unclear; each row says why.",
      "rows": [
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Input)",
          "aValue": 0.037,
          "bValue": 0.05,
          "unit": "usd",
          "aDisplay": "$0.037",
          "bDisplay": "$0.050",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.037 vs $0.050, 1.4x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "bf16 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Output)",
          "aValue": 0.17,
          "bValue": 0.25,
          "unit": "usd",
          "aDisplay": "$0.17",
          "bDisplay": "$0.25",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.17 vs $0.25, 1.5x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "bf16 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Llama 3.3 70B Instruct: price per million tokens by provider (Input)",
          "aValue": 0.1,
          "bValue": 0.135,
          "unit": "usd",
          "aDisplay": "$0.10",
          "bDisplay": "$0.14",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.10 vs $0.14, 1.4x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-llama-3-3-70b-instruct",
          "aContext": "fp8 · Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "bf16 · Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Llama 3.3 70B Instruct: price per million tokens by provider (Output)",
          "aValue": 0.32,
          "bValue": 0.4,
          "unit": "usd",
          "aDisplay": "$0.32",
          "bDisplay": "$0.40",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.32 vs $0.40, 1.3x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-llama-3-3-70b-instruct",
          "aContext": "fp8 · Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "bf16 · Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "DeepSeek V4 Pro 0423: price per million tokens by provider (Input)",
          "aValue": 1.3,
          "bValue": 1.6,
          "unit": "usd",
          "aDisplay": "$1.30",
          "bDisplay": "$1.60",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($1.30 vs $1.60, 1.2x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-deepseek-v4-pro",
          "aContext": "fp8 · DeepSeek V4 Pro 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · DeepSeek V4 Pro 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "DeepSeek V4 Pro 0423: price per million tokens by provider (Output)",
          "aValue": 2.6,
          "bValue": 3.2,
          "unit": "usd",
          "aDisplay": "$2.60",
          "bDisplay": "$3.20",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($2.60 vs $3.20, 1.2x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-deepseek-v4-pro",
          "aContext": "fp8 · DeepSeek V4 Pro 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · DeepSeek V4 Pro 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "DeepSeek V4 Pro 0423: price per million tokens by provider (Cache read)",
          "aValue": 0.1,
          "bValue": 0.135,
          "unit": "usd",
          "aDisplay": "$0.10",
          "bDisplay": "$0.14",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.10 vs $0.14, 1.4x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-deepseek-v4-pro",
          "aContext": "fp8 · DeepSeek V4 Pro 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · DeepSeek V4 Pro 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "DeepSeek V4 Flash 0423: price per million tokens by provider (Input)",
          "aValue": 0.09,
          "bValue": 0.14,
          "unit": "usd",
          "aDisplay": "$0.090",
          "bDisplay": "$0.14",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.090 vs $0.14, 1.6x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-deepseek-v4-flash",
          "aContext": "fp8 · DeepSeek V4 Flash 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · DeepSeek V4 Flash 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "DeepSeek V4 Flash 0423: price per million tokens by provider (Output)",
          "aValue": 0.18,
          "bValue": 0.28,
          "unit": "usd",
          "aDisplay": "$0.18",
          "bDisplay": "$0.28",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.18 vs $0.28, 1.6x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-deepseek-v4-flash",
          "aContext": "fp8 · DeepSeek V4 Flash 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · DeepSeek V4 Flash 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "DeepSeek V4 Flash 0423: price per million tokens by provider (Cache read)",
          "aValue": 0.018,
          "bValue": 0.028,
          "unit": "usd",
          "aDisplay": "$0.018",
          "bDisplay": "$0.028",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.018 vs $0.028, 1.6x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-deepseek-v4-flash",
          "aContext": "fp8 · DeepSeek V4 Flash 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · DeepSeek V4 Flash 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "GLM 5.3: price per million tokens by provider (Input)",
          "aValue": 0.5625,
          "bValue": 0.42,
          "unit": "usd",
          "aDisplay": "$0.56",
          "bDisplay": "$0.42",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.56 vs $0.42, 1.3x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "aContext": "fp4 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "GLM 5.3: price per million tokens by provider (Output)",
          "aValue": 2.5,
          "bValue": 1.32,
          "unit": "usd",
          "aDisplay": "$2.50",
          "bDisplay": "$1.32",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($2.50 vs $1.32, 1.9x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "aContext": "fp4 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "GLM 5.3: price per million tokens by provider (Cache read)",
          "aValue": 0.125,
          "bValue": 0.078,
          "unit": "usd",
          "aDisplay": "$0.13",
          "bDisplay": "$0.078",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.13 vs $0.078, 1.6x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "aContext": "fp4 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "claude-haiku-4-5-vs-claude-fable-5-1",
      "a": "claude-haiku-4-5",
      "b": "claude-fable-5-1",
      "title": "Claude Haiku 4.5 vs Claude Fable 5.1",
      "seoTitle": "Claude Haiku 4.5 vs Claude Fable 5.1: measured benchmarks",
      "description": "Claude Haiku 4.5 vs Claude Fable 5.1: 12 measured metrics from 3 studies (Pass rate on five validated tasks; more), with sample sizes and intervals.",
      "verdict": "Claude Haiku 4.5 and Claude Fable 5.1 share 12 measured metrics and 11 list-price calculations from 4 studies. Claude Fable 5.1 leads on 2 rows: Pass rate on eight hard tasks (Strict pass), 100% (24/24) vs 46% (11/24); Pass rate on eight hard tasks (Lenient (format misses counted)), 100% (24/24) vs 67% (16/24). On those rows the 95% intervals do not overlap. The other rows are 1 tie and 20 unclear; each row says why. Calculation rows are derived from list prices and recorded counts; they are not bills or runs. Some rows rest on small samples (n = 3 at the smallest).",
      "rows": [
        {
          "metric": "Pass rate on five validated tasks",
          "aValue": 1,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "100% (15/15)",
          "bDisplay": "100% (15/15)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Haiku 4.5 80% to 100%; Claude Fable 5.1 80% to 100%), so this sample cannot separate them.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-pass-rate",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code · five short validated tasks",
          "bContext": "Claude Code · five short validated tasks",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.7961,
            1
          ],
          "bRange": [
            0.7961,
            1
          ]
        },
        {
          "metric": "Total time per call",
          "aValue": 4.43,
          "bValue": 1.94,
          "unit": "seconds",
          "aDisplay": "4.43 s",
          "bDisplay": "1.94 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Haiku 4.5 3.16 s to 23.6 s; Claude Fable 5.1 1.41 s to 9.83 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-total-latency",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code · five short validated tasks",
          "bContext": "Claude Code · five short validated tasks",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            3.16,
            23.57
          ],
          "bRange": [
            1.41,
            9.83
          ]
        },
        {
          "metric": "Time to first useful output",
          "aValue": 3.63,
          "bValue": 1.2,
          "unit": "seconds",
          "aDisplay": "3.63 s",
          "bDisplay": "1.20 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Haiku 4.5 2.78 s to 22.3 s; Claude Fable 5.1 0.95 s to 7.90 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-first-useful-latency",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code · five short validated tasks",
          "bContext": "Claude Code · five short validated tasks",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            2.78,
            22.27
          ],
          "bRange": [
            0.95,
            7.9
          ]
        },
        {
          "metric": "Input tokens per call: what the CLI sends (Cache read)",
          "aValue": 0,
          "bValue": 2760,
          "unit": "tokens",
          "aDisplay": "0",
          "bDisplay": "2,760",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-input-tokens",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code · five short validated tasks",
          "bContext": "Claude Code · five short validated tasks"
        },
        {
          "metric": "Input tokens per call: what the CLI sends (Other input)",
          "aValue": 3790,
          "bValue": 473,
          "unit": "tokens",
          "aDisplay": "3,790",
          "bDisplay": "473",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-input-tokens",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code · five short validated tasks",
          "bContext": "Claude Code · five short validated tasks"
        },
        {
          "metric": "Output tokens per call (Output tokens)",
          "aValue": 367,
          "bValue": 64,
          "unit": "tokens",
          "aDisplay": "367",
          "bDisplay": "64",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-output-tokens",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code · five short validated tasks",
          "bContext": "Claude Code · five short validated tasks"
        },
        {
          "metric": "List-price cost per call (calculation)",
          "aValue": 0.00566,
          "bValue": 0.00987,
          "unit": "usd",
          "aDisplay": "$0.0057",
          "bDisplay": "$0.0099",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Haiku 4.5 $0.0051 to $0.018; Claude Fable 5.1 $0.0049 to $0.058); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-list-price-per-call",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code · five short validated tasks",
          "bContext": "Claude Code · five short validated tasks",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            0.00513,
            0.01804
          ],
          "bRange": [
            0.0049,
            0.05843
          ],
          "calculation": true
        },
        {
          "metric": "List-price cost per passing answer (calculation)",
          "aValue": 0.00836,
          "bValue": 0.02054,
          "unit": "usd",
          "aDisplay": "$0.0084",
          "bDisplay": "$0.021",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.0084 vs $0.021, 2.5x) is not tested against run-to-run variation.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-cost-per-pass",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code · five short validated tasks",
          "bContext": "Claude Code · five short validated tasks",
          "calculation": true
        },
        {
          "metric": "Pass rate on eight hard tasks (Strict pass)",
          "aValue": 0.4583,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "46% (11/24)",
          "bDisplay": "100% (24/24)",
          "winner": "b",
          "basis": "The 95% intervals do not overlap (Claude Haiku 4.5 28% to 65%; Claude Fable 5.1 86% to 100%).",
          "studySlug": "hard-model-head-to-head",
          "n": 24,
          "chartId": "hard-h2h-pass-rate",
          "aN": 24,
          "bN": 24,
          "aContext": "Claude Code · eight hard validated tasks",
          "bContext": "Claude Code · eight hard validated tasks",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.2789,
            0.6493
          ],
          "bRange": [
            0.862,
            1
          ]
        },
        {
          "metric": "Pass rate on eight hard tasks (Lenient (format misses counted))",
          "aValue": 0.6667,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "67% (16/24)",
          "bDisplay": "100% (24/24)",
          "winner": "b",
          "basis": "The 95% intervals do not overlap (Claude Haiku 4.5 47% to 82%; Claude Fable 5.1 86% to 100%).",
          "studySlug": "hard-model-head-to-head",
          "n": 24,
          "chartId": "hard-h2h-pass-rate",
          "aN": 24,
          "bN": 24,
          "aContext": "Claude Code · eight hard validated tasks",
          "bContext": "Claude Code · eight hard validated tasks",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.4671,
            0.8203
          ],
          "bRange": [
            0.862,
            1
          ]
        },
        {
          "metric": "Total time per call on hard tasks (separate batches)",
          "aValue": 39.01,
          "bValue": 16.13,
          "unit": "seconds",
          "aDisplay": "39.0 s",
          "bDisplay": "16.1 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Haiku 4.5 15.3 s to 75.1 s; Claude Fable 5.1 4.46 s to 90.0 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "hard-model-head-to-head",
          "n": 24,
          "chartId": "hard-h2h-total-latency",
          "aN": 24,
          "bN": 24,
          "aContext": "Claude Code · eight hard validated tasks",
          "bContext": "Claude Code · eight hard validated tasks",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            15.27,
            75.13
          ],
          "bRange": [
            4.46,
            90
          ]
        },
        {
          "metric": "Time to first useful output on hard tasks",
          "aValue": 35.54,
          "bValue": 11.63,
          "unit": "seconds",
          "aDisplay": "35.5 s",
          "bDisplay": "11.6 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Haiku 4.5 12.9 s to 70.3 s; Claude Fable 5.1 2.00 s to 85.3 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "hard-model-head-to-head",
          "n": 24,
          "chartId": "hard-h2h-first-useful-latency",
          "aN": 24,
          "bN": 24,
          "aContext": "Claude Code · eight hard validated tasks",
          "bContext": "Claude Code · eight hard validated tasks",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            12.88,
            70.31
          ],
          "bRange": [
            2,
            85.33
          ]
        },
        {
          "metric": "Output tokens per call on hard tasks (Output tokens)",
          "aValue": 5064,
          "bValue": 1366,
          "unit": "tokens",
          "aDisplay": "5,064",
          "bDisplay": "1,366",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "hard-model-head-to-head",
          "n": 24,
          "chartId": "hard-h2h-output-tokens",
          "aN": 24,
          "bN": 24,
          "aContext": "Claude Code · eight hard validated tasks",
          "bContext": "Claude Code · eight hard validated tasks"
        },
        {
          "metric": "List-price cost per strict pass on hard tasks (calculation)",
          "aValue": 0.0672,
          "bValue": 0.09331,
          "unit": "usd",
          "aDisplay": "$0.067",
          "bDisplay": "$0.093",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.067 vs $0.093) is not tested against run-to-run variation.",
          "studySlug": "hard-model-head-to-head",
          "n": 24,
          "chartId": "hard-h2h-cost-per-pass",
          "aN": 24,
          "bN": 24,
          "aContext": "Claude Code · eight hard validated tasks",
          "bContext": "Claude Code · eight hard validated tasks",
          "calculation": true
        },
        {
          "metric": "Reasoning share of output tokens per call on hard tasks (calculation)",
          "aValue": 91.68,
          "bValue": 64.24,
          "unit": "percent",
          "aDisplay": "91.7%",
          "bDisplay": "64.2%",
          "winner": "unclear",
          "basis": "More or fewer percent is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "n": 24,
          "chartId": "thinking-bill-share",
          "aN": 24,
          "bN": 24,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            76.46,
            99.27
          ],
          "bRange": [
            23.44,
            97.19
          ],
          "calculation": true
        },
        {
          "metric": "List-price cost per call: reasoning, remaining output and input (calculation) (Reasoning (output tokens))",
          "aValue": 0.024492,
          "bValue": 0.053696,
          "unit": "usd",
          "aDisplay": "$0.024",
          "bDisplay": "$0.054",
          "winner": "unclear",
          "basis": "More or fewer usd is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "n": 24,
          "chartId": "thinking-bill-cost-per-call",
          "aN": 24,
          "bN": 24,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "calculation": true
        },
        {
          "metric": "List-price cost per call: reasoning, remaining output and input (calculation) (Remaining output (visible-answer estimate))",
          "aValue": 0.001795,
          "bValue": 0.018702,
          "unit": "usd",
          "aDisplay": "$0.0018",
          "bDisplay": "$0.019",
          "winner": "unclear",
          "basis": "More or fewer usd is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "n": 24,
          "chartId": "thinking-bill-cost-per-call",
          "aN": 24,
          "bN": 24,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "calculation": true
        },
        {
          "metric": "List-price cost per call: reasoning, remaining output and input (calculation) (Input (prompt, cache priced))",
          "aValue": 0.00451,
          "bValue": 0.02091,
          "unit": "usd",
          "aDisplay": "$0.0045",
          "bDisplay": "$0.021",
          "winner": "unclear",
          "basis": "More or fewer usd is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "n": 24,
          "chartId": "thinking-bill-cost-per-call",
          "aN": 24,
          "bN": 24,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "calculation": true
        },
        {
          "metric": "Reasoning share on short tasks vs hard tasks (calculation) (Eight hard tasks)",
          "aValue": 91.68,
          "bValue": 64.24,
          "unit": "percent",
          "aDisplay": "91.7%",
          "bDisplay": "64.2%",
          "winner": "unclear",
          "basis": "More or fewer percent is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "n": 24,
          "chartId": "thinking-bill-short-vs-hard",
          "aN": 24,
          "bN": 24,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            76.46,
            99.27
          ],
          "bRange": [
            23.44,
            97.19
          ],
          "calculation": true
        },
        {
          "metric": "Reasoning share on short tasks vs hard tasks (calculation) (Five short tasks)",
          "aValue": 90.19,
          "bValue": 0,
          "unit": "percent",
          "aDisplay": "90.2%",
          "bDisplay": "0%",
          "winner": "unclear",
          "basis": "More or fewer percent is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "n": 15,
          "chartId": "thinking-bill-short-vs-hard",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            73.1,
            97.59
          ],
          "bRange": [
            0,
            74.01
          ],
          "calculation": true
        },
        {
          "metric": "Time to first text: a 250-line answer, six models",
          "aValue": 4,
          "bValue": 4.43,
          "unit": "seconds",
          "aDisplay": "4.00 s",
          "bDisplay": "4.43 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Haiku 4.5 2.84 s to 6.38 s; Claude Fable 5.1 2.27 s to 4.64 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "llm-speed-anatomy",
          "n": 4,
          "chartId": "speed-anatomy-first-text",
          "aN": 4,
          "bN": 4,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            2.84,
            6.38
          ],
          "bRange": [
            2.27,
            4.64
          ]
        },
        {
          "metric": "Output speed after the first text: visible tokens per second (calculation)",
          "aValue": 153.2,
          "bValue": 122.6,
          "unit": "tokens",
          "aDisplay": "153",
          "bDisplay": "123",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "llm-speed-anatomy",
          "n": 4,
          "chartId": "speed-anatomy-output-speed",
          "aN": 4,
          "bN": 4,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            152.6,
            216.1
          ],
          "bRange": [
            120.9,
            131.4
          ],
          "calculation": true
        },
        {
          "metric": "Output speed in characters per second after the first text (calculation)",
          "aValue": 547,
          "bValue": 273,
          "unit": "count",
          "aDisplay": "547",
          "bDisplay": "273",
          "winner": "unclear",
          "basis": "More or fewer count is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "llm-speed-anatomy",
          "chartId": "speed-anatomy-chars-per-second",
          "aN": 3,
          "bN": 4,
          "aContext": "Claude Code",
          "bContext": "Claude Code",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            546,
            548
          ],
          "bRange": [
            270,
            293
          ],
          "calculation": true
        }
      ]
    },
    {
      "slug": "google-ai-studio-vs-google-vertex",
      "a": "google-ai-studio",
      "b": "google-vertex",
      "title": "Google AI Studio vs Google Vertex AI",
      "seoTitle": "Google AI Studio vs Google Vertex AI: price per model",
      "description": "Google AI Studio vs Google Vertex AI: 12 list prices per million tokens for the same models, as reported by OpenRouter’s public API, with the gap stated.",
      "verdict": "Google AI Studio and Google Vertex AI share 12 reported list prices from one study. No row names a winner: a reported list price has no interval, so each gap is stated, not ranked. Prices change often; the snapshot date is in each row’s context. The rows are 12 ties; each row says why.",
      "rows": [
        {
          "metric": "Gemini 3.8 Flash: price per million tokens by provider (Input)",
          "aValue": 0.75,
          "bValue": 0.75,
          "unit": "usd",
          "aDisplay": "$0.75",
          "bDisplay": "$0.75",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gemini-3-8-flash",
          "aContext": "Gemini 3.8 Flash · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Gemini 3.8 Flash · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Gemini 3.8 Flash: price per million tokens by provider (Output)",
          "aValue": 3.75,
          "bValue": 3.75,
          "unit": "usd",
          "aDisplay": "$3.75",
          "bDisplay": "$3.75",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gemini-3-8-flash",
          "aContext": "Gemini 3.8 Flash · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Gemini 3.8 Flash · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Gemini 3.8 Flash: price per million tokens by provider (Cache read)",
          "aValue": 0.075,
          "bValue": 0.075,
          "unit": "usd",
          "aDisplay": "$0.075",
          "bDisplay": "$0.075",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gemini-3-8-flash",
          "aContext": "Gemini 3.8 Flash · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Gemini 3.8 Flash · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Gemini 3.5 Flash: price per million tokens by provider (Input)",
          "aValue": 1.5,
          "bValue": 1.5,
          "unit": "usd",
          "aDisplay": "$1.50",
          "bDisplay": "$1.50",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gemini-3-5-flash",
          "aContext": "Gemini 3.5 Flash · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Gemini 3.5 Flash · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Gemini 3.5 Flash: price per million tokens by provider (Output)",
          "aValue": 9,
          "bValue": 9,
          "unit": "usd",
          "aDisplay": "$9.00",
          "bDisplay": "$9.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gemini-3-5-flash",
          "aContext": "Gemini 3.5 Flash · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Gemini 3.5 Flash · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Gemini 3.5 Flash: price per million tokens by provider (Cache read)",
          "aValue": 0.15,
          "bValue": 0.15,
          "unit": "usd",
          "aDisplay": "$0.15",
          "bDisplay": "$0.15",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gemini-3-5-flash",
          "aContext": "Gemini 3.5 Flash · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Gemini 3.5 Flash · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Gemini 3.5 Flash Lite: price per million tokens by provider (Input)",
          "aValue": 0.3,
          "bValue": 0.3,
          "unit": "usd",
          "aDisplay": "$0.30",
          "bDisplay": "$0.30",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gemini-3-5-flash-lite",
          "aContext": "Gemini 3.5 Flash Lite · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Gemini 3.5 Flash Lite · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Gemini 3.5 Flash Lite: price per million tokens by provider (Output)",
          "aValue": 2.5,
          "bValue": 2.5,
          "unit": "usd",
          "aDisplay": "$2.50",
          "bDisplay": "$2.50",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gemini-3-5-flash-lite",
          "aContext": "Gemini 3.5 Flash Lite · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Gemini 3.5 Flash Lite · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Gemini 3.5 Flash Lite: price per million tokens by provider (Cache read)",
          "aValue": 0.03,
          "bValue": 0.03,
          "unit": "usd",
          "aDisplay": "$0.030",
          "bDisplay": "$0.030",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gemini-3-5-flash-lite",
          "aContext": "Gemini 3.5 Flash Lite · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Gemini 3.5 Flash Lite · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Gemini 3.1 Pro Preview: price per million tokens by provider (Input)",
          "aValue": 2,
          "bValue": 2,
          "unit": "usd",
          "aDisplay": "$2.00",
          "bDisplay": "$2.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gemini-3-1-pro-preview",
          "aContext": "Gemini 3.1 Pro Preview · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Gemini 3.1 Pro Preview · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Gemini 3.1 Pro Preview: price per million tokens by provider (Output)",
          "aValue": 12,
          "bValue": 12,
          "unit": "usd",
          "aDisplay": "$12.00",
          "bDisplay": "$12.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gemini-3-1-pro-preview",
          "aContext": "Gemini 3.1 Pro Preview · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Gemini 3.1 Pro Preview · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Gemini 3.1 Pro Preview: price per million tokens by provider (Cache read)",
          "aValue": 0.2,
          "bValue": 0.2,
          "unit": "usd",
          "aDisplay": "$0.20",
          "bDisplay": "$0.20",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gemini-3-1-pro-preview",
          "aContext": "Gemini 3.1 Pro Preview · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Gemini 3.1 Pro Preview · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "openai-vs-azure",
      "a": "openai",
      "b": "azure",
      "title": "OpenAI vs Azure",
      "seoTitle": "OpenAI vs Azure: price per model",
      "description": "OpenAI vs Azure: 12 list prices per million tokens for the same models, as reported by OpenRouter’s public API, with the gap stated.",
      "verdict": "OpenAI and Azure share 12 reported list prices from one study. No row names a winner: a reported list price has no interval, so each gap is stated, not ranked. Prices change often; the snapshot date is in each row’s context. The rows are 12 ties; each row says why.",
      "rows": [
        {
          "metric": "GPT-6 Sol: price per million tokens by provider (Input)",
          "aValue": 2,
          "bValue": 2,
          "unit": "usd",
          "aDisplay": "$2.00",
          "bDisplay": "$2.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-6-sol",
          "aContext": "GPT-6 Sol · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "GPT-6 Sol · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "GPT-6 Sol: price per million tokens by provider (Output)",
          "aValue": 10,
          "bValue": 10,
          "unit": "usd",
          "aDisplay": "$10.00",
          "bDisplay": "$10.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-6-sol",
          "aContext": "GPT-6 Sol · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "GPT-6 Sol · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "GPT-6 Sol: price per million tokens by provider (Cache read)",
          "aValue": 0.2,
          "bValue": 0.2,
          "unit": "usd",
          "aDisplay": "$0.20",
          "bDisplay": "$0.20",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-6-sol",
          "aContext": "GPT-6 Sol · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "GPT-6 Sol · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "GPT-6 Luna: price per million tokens by provider (Input)",
          "aValue": 0.1,
          "bValue": 0.1,
          "unit": "usd",
          "aDisplay": "$0.10",
          "bDisplay": "$0.10",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-6-luna",
          "aContext": "GPT-6 Luna · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "GPT-6 Luna · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "GPT-6 Luna: price per million tokens by provider (Output)",
          "aValue": 0.5,
          "bValue": 0.5,
          "unit": "usd",
          "aDisplay": "$0.50",
          "bDisplay": "$0.50",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-6-luna",
          "aContext": "GPT-6 Luna · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "GPT-6 Luna · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "GPT-6 Luna: price per million tokens by provider (Cache read)",
          "aValue": 0.01,
          "bValue": 0.01,
          "unit": "usd",
          "aDisplay": "$0.010",
          "bDisplay": "$0.010",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-6-luna",
          "aContext": "GPT-6 Luna · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "GPT-6 Luna · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "GPT-6 Astra: price per million tokens by provider (Input)",
          "aValue": 10,
          "bValue": 10,
          "unit": "usd",
          "aDisplay": "$10.00",
          "bDisplay": "$10.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-6-astra",
          "aContext": "GPT-6 Astra · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "GPT-6 Astra · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "GPT-6 Astra: price per million tokens by provider (Output)",
          "aValue": 50,
          "bValue": 50,
          "unit": "usd",
          "aDisplay": "$50.00",
          "bDisplay": "$50.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-6-astra",
          "aContext": "GPT-6 Astra · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "GPT-6 Astra · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "GPT-6 Astra: price per million tokens by provider (Cache read)",
          "aValue": 1,
          "bValue": 1,
          "unit": "usd",
          "aDisplay": "$1.00",
          "bDisplay": "$1.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-6-astra",
          "aContext": "GPT-6 Astra · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "GPT-6 Astra · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "GPT-5.5: price per million tokens by provider (Input)",
          "aValue": 5,
          "bValue": 5,
          "unit": "usd",
          "aDisplay": "$5.00",
          "bDisplay": "$5.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-5-5",
          "aContext": "GPT-5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "GPT-5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "GPT-5.5: price per million tokens by provider (Output)",
          "aValue": 30,
          "bValue": 30,
          "unit": "usd",
          "aDisplay": "$30.00",
          "bDisplay": "$30.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-5-5",
          "aContext": "GPT-5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "GPT-5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "GPT-5.5: price per million tokens by provider (Cache read)",
          "aValue": 0.5,
          "bValue": 0.5,
          "unit": "usd",
          "aDisplay": "$0.50",
          "bDisplay": "$0.50",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-5-5",
          "aContext": "GPT-5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "GPT-5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "parasail-vs-siliconflow",
      "a": "parasail",
      "b": "siliconflow",
      "title": "Parasail vs SiliconFlow",
      "seoTitle": "Parasail vs SiliconFlow: price per model",
      "description": "Parasail vs SiliconFlow: 12 list prices per million tokens for the same models, as reported by OpenRouter’s public API, with the gap stated.",
      "verdict": "Parasail and SiliconFlow share 12 reported list prices from one study. No row names a winner: a reported list price has no interval, so each gap is stated, not ranked. Prices change often; the snapshot date is in each row’s context. The rows are 1 tie and 11 unclear; each row says why.",
      "rows": [
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Input)",
          "aValue": 0.1,
          "bValue": 0.15,
          "unit": "usd",
          "aDisplay": "$0.10",
          "bDisplay": "$0.15",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.10 vs $0.15, 1.5x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Output)",
          "aValue": 0.75,
          "bValue": 0.6,
          "unit": "usd",
          "aDisplay": "$0.75",
          "bDisplay": "$0.60",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.75 vs $0.60, 1.3x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Cache read)",
          "aValue": 0.055,
          "bValue": 0.075,
          "unit": "usd",
          "aDisplay": "$0.055",
          "bDisplay": "$0.075",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.055 vs $0.075, 1.4x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "DeepSeek V4 Pro 0423: price per million tokens by provider (Input)",
          "aValue": 0.45,
          "bValue": 1.50162,
          "unit": "usd",
          "aDisplay": "$0.45",
          "bDisplay": "$1.50",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.45 vs $1.50, 3.3x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-deepseek-v4-pro",
          "aContext": "fp8 · DeepSeek V4 Pro 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · DeepSeek V4 Pro 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "DeepSeek V4 Pro 0423: price per million tokens by provider (Output)",
          "aValue": 3.48,
          "bValue": 3.135,
          "unit": "usd",
          "aDisplay": "$3.48",
          "bDisplay": "$3.13",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($3.48 vs $3.13, 1.1x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-deepseek-v4-pro",
          "aContext": "fp8 · DeepSeek V4 Pro 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · DeepSeek V4 Pro 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "DeepSeek V4 Pro 0423: price per million tokens by provider (Cache read)",
          "aValue": 0.1,
          "bValue": 0.135,
          "unit": "usd",
          "aDisplay": "$0.10",
          "bDisplay": "$0.14",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.10 vs $0.14, 1.4x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-deepseek-v4-pro",
          "aContext": "fp8 · DeepSeek V4 Pro 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · DeepSeek V4 Pro 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "DeepSeek V4 Flash 0423: price per million tokens by provider (Input)",
          "aValue": 0.14,
          "bValue": 0.13,
          "unit": "usd",
          "aDisplay": "$0.14",
          "bDisplay": "$0.13",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.14 vs $0.13, 1.1x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-deepseek-v4-flash",
          "aContext": "fp8 · DeepSeek V4 Flash 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · DeepSeek V4 Flash 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "DeepSeek V4 Flash 0423: price per million tokens by provider (Output)",
          "aValue": 0.28,
          "bValue": 0.28,
          "unit": "usd",
          "aDisplay": "$0.28",
          "bDisplay": "$0.28",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-deepseek-v4-flash",
          "aContext": "fp8 · DeepSeek V4 Flash 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · DeepSeek V4 Flash 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "DeepSeek V4 Flash 0423: price per million tokens by provider (Cache read)",
          "aValue": 0.07,
          "bValue": 0.028,
          "unit": "usd",
          "aDisplay": "$0.070",
          "bDisplay": "$0.028",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.070 vs $0.028, 2.5x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-deepseek-v4-flash",
          "aContext": "fp8 · DeepSeek V4 Flash 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · DeepSeek V4 Flash 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "GLM 5.3: price per million tokens by provider (Input)",
          "aValue": 1.4,
          "bValue": 0.7,
          "unit": "usd",
          "aDisplay": "$1.40",
          "bDisplay": "$0.70",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($1.40 vs $0.70, 2.0x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "aContext": "fp8 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "GLM 5.3: price per million tokens by provider (Output)",
          "aValue": 4.4,
          "bValue": 2.2,
          "unit": "usd",
          "aDisplay": "$4.40",
          "bDisplay": "$2.20",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($4.40 vs $2.20, 2.0x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "aContext": "fp8 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "GLM 5.3: price per million tokens by provider (Cache read)",
          "aValue": 0.26,
          "bValue": 0.13,
          "unit": "usd",
          "aDisplay": "$0.26",
          "bDisplay": "$0.13",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.26 vs $0.13, 2.0x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "aContext": "fp8 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "deepinfra-vs-cloudflare",
      "a": "deepinfra",
      "b": "cloudflare",
      "title": "DeepInfra vs Cloudflare Workers AI",
      "seoTitle": "DeepInfra vs Cloudflare Workers AI: price per model",
      "description": "DeepInfra vs Cloudflare Workers AI: 11 list prices per million tokens for the same models, as reported by OpenRouter’s public API, with the gap stated.",
      "verdict": "DeepInfra and Cloudflare Workers AI share 11 reported list prices from one study. No row names a winner: a reported list price has no interval, so each gap is stated, not ranked. Prices change often; the snapshot date is in each row’s context. The rows are 11 unclear; each row says why.",
      "rows": [
        {
          "metric": "Llama 3.3 70B Instruct: price per million tokens by provider (Input)",
          "aValue": 0.1,
          "bValue": 0.293,
          "unit": "usd",
          "aDisplay": "$0.10",
          "bDisplay": "$0.29",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.10 vs $0.29, 2.9x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-llama-3-3-70b-instruct",
          "aContext": "fp8 · Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Llama 3.3 70B Instruct: price per million tokens by provider (Output)",
          "aValue": 0.32,
          "bValue": 2.253,
          "unit": "usd",
          "aDisplay": "$0.32",
          "bDisplay": "$2.25",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.32 vs $2.25, 7.0x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-llama-3-3-70b-instruct",
          "aContext": "fp8 · Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "DeepSeek V4 Pro 0423: price per million tokens by provider (Input)",
          "aValue": 1.3,
          "bValue": 1.15,
          "unit": "usd",
          "aDisplay": "$1.30",
          "bDisplay": "$1.15",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($1.30 vs $1.15, 1.1x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-deepseek-v4-pro",
          "aContext": "fp8 · DeepSeek V4 Pro 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "DeepSeek V4 Pro 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "DeepSeek V4 Pro 0423: price per million tokens by provider (Output)",
          "aValue": 2.6,
          "bValue": 2.55,
          "unit": "usd",
          "aDisplay": "$2.60",
          "bDisplay": "$2.55",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($2.60 vs $2.55, 1.0x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-deepseek-v4-pro",
          "aContext": "fp8 · DeepSeek V4 Pro 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "DeepSeek V4 Pro 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "DeepSeek V4 Pro 0423: price per million tokens by provider (Cache read)",
          "aValue": 0.1,
          "bValue": 0.2,
          "unit": "usd",
          "aDisplay": "$0.10",
          "bDisplay": "$0.20",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.10 vs $0.20, 2.0x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-deepseek-v4-pro",
          "aContext": "fp8 · DeepSeek V4 Pro 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "DeepSeek V4 Pro 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "DeepSeek V4 Flash 0423: price per million tokens by provider (Input)",
          "aValue": 0.09,
          "bValue": 0.44,
          "unit": "usd",
          "aDisplay": "$0.090",
          "bDisplay": "$0.44",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.090 vs $0.44, 4.9x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-deepseek-v4-flash",
          "aContext": "fp8 · DeepSeek V4 Flash 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "DeepSeek V4 Flash 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "DeepSeek V4 Flash 0423: price per million tokens by provider (Output)",
          "aValue": 0.18,
          "bValue": 1.32,
          "unit": "usd",
          "aDisplay": "$0.18",
          "bDisplay": "$1.32",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.18 vs $1.32, 7.3x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-deepseek-v4-flash",
          "aContext": "fp8 · DeepSeek V4 Flash 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "DeepSeek V4 Flash 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "DeepSeek V4 Flash 0423: price per million tokens by provider (Cache read)",
          "aValue": 0.018,
          "bValue": 0.014,
          "unit": "usd",
          "aDisplay": "$0.018",
          "bDisplay": "$0.014",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.018 vs $0.014, 1.3x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-deepseek-v4-flash",
          "aContext": "fp8 · DeepSeek V4 Flash 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "DeepSeek V4 Flash 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "GLM 5.3: price per million tokens by provider (Input)",
          "aValue": 0.5625,
          "bValue": 1.4,
          "unit": "usd",
          "aDisplay": "$0.56",
          "bDisplay": "$1.40",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.56 vs $1.40, 2.5x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "aContext": "fp4 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "GLM 5.3: price per million tokens by provider (Output)",
          "aValue": 2.5,
          "bValue": 4.4,
          "unit": "usd",
          "aDisplay": "$2.50",
          "bDisplay": "$4.40",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($2.50 vs $4.40, 1.8x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "aContext": "fp4 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "GLM 5.3: price per million tokens by provider (Cache read)",
          "aValue": 0.125,
          "bValue": 0.26,
          "unit": "usd",
          "aDisplay": "$0.13",
          "bDisplay": "$0.26",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.13 vs $0.26, 2.1x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "aContext": "fp4 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "deepinfra-vs-siliconflow",
      "a": "deepinfra",
      "b": "siliconflow",
      "title": "DeepInfra vs SiliconFlow",
      "seoTitle": "DeepInfra vs SiliconFlow: price per model",
      "description": "DeepInfra vs SiliconFlow: 11 list prices per million tokens for the same models, as reported by OpenRouter’s public API, with the gap stated.",
      "verdict": "DeepInfra and SiliconFlow share 11 reported list prices from one study. No row names a winner: a reported list price has no interval, so each gap is stated, not ranked. Prices change often; the snapshot date is in each row’s context. The rows are 11 unclear; each row says why.",
      "rows": [
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Input)",
          "aValue": 0.037,
          "bValue": 0.15,
          "unit": "usd",
          "aDisplay": "$0.037",
          "bDisplay": "$0.15",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.037 vs $0.15, 4.1x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "bf16 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Output)",
          "aValue": 0.17,
          "bValue": 0.6,
          "unit": "usd",
          "aDisplay": "$0.17",
          "bDisplay": "$0.60",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.17 vs $0.60, 3.5x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "bf16 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "DeepSeek V4 Pro 0423: price per million tokens by provider (Input)",
          "aValue": 1.3,
          "bValue": 1.50162,
          "unit": "usd",
          "aDisplay": "$1.30",
          "bDisplay": "$1.50",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($1.30 vs $1.50, 1.2x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-deepseek-v4-pro",
          "aContext": "fp8 · DeepSeek V4 Pro 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · DeepSeek V4 Pro 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "DeepSeek V4 Pro 0423: price per million tokens by provider (Output)",
          "aValue": 2.6,
          "bValue": 3.135,
          "unit": "usd",
          "aDisplay": "$2.60",
          "bDisplay": "$3.13",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($2.60 vs $3.13, 1.2x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-deepseek-v4-pro",
          "aContext": "fp8 · DeepSeek V4 Pro 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · DeepSeek V4 Pro 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "DeepSeek V4 Pro 0423: price per million tokens by provider (Cache read)",
          "aValue": 0.1,
          "bValue": 0.135,
          "unit": "usd",
          "aDisplay": "$0.10",
          "bDisplay": "$0.14",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.10 vs $0.14, 1.4x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-deepseek-v4-pro",
          "aContext": "fp8 · DeepSeek V4 Pro 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · DeepSeek V4 Pro 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "DeepSeek V4 Flash 0423: price per million tokens by provider (Input)",
          "aValue": 0.09,
          "bValue": 0.13,
          "unit": "usd",
          "aDisplay": "$0.090",
          "bDisplay": "$0.13",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.090 vs $0.13, 1.4x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-deepseek-v4-flash",
          "aContext": "fp8 · DeepSeek V4 Flash 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · DeepSeek V4 Flash 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "DeepSeek V4 Flash 0423: price per million tokens by provider (Output)",
          "aValue": 0.18,
          "bValue": 0.28,
          "unit": "usd",
          "aDisplay": "$0.18",
          "bDisplay": "$0.28",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.18 vs $0.28, 1.6x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-deepseek-v4-flash",
          "aContext": "fp8 · DeepSeek V4 Flash 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · DeepSeek V4 Flash 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "DeepSeek V4 Flash 0423: price per million tokens by provider (Cache read)",
          "aValue": 0.018,
          "bValue": 0.028,
          "unit": "usd",
          "aDisplay": "$0.018",
          "bDisplay": "$0.028",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.018 vs $0.028, 1.6x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-deepseek-v4-flash",
          "aContext": "fp8 · DeepSeek V4 Flash 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · DeepSeek V4 Flash 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "GLM 5.3: price per million tokens by provider (Input)",
          "aValue": 0.5625,
          "bValue": 0.7,
          "unit": "usd",
          "aDisplay": "$0.56",
          "bDisplay": "$0.70",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.56 vs $0.70, 1.2x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "aContext": "fp4 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "GLM 5.3: price per million tokens by provider (Output)",
          "aValue": 2.5,
          "bValue": 2.2,
          "unit": "usd",
          "aDisplay": "$2.50",
          "bDisplay": "$2.20",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($2.50 vs $2.20, 1.1x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "aContext": "fp4 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "GLM 5.3: price per million tokens by provider (Cache read)",
          "aValue": 0.125,
          "bValue": 0.13,
          "unit": "usd",
          "aDisplay": "$0.13",
          "bDisplay": "$0.13",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.13 vs $0.13, 1.0x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "aContext": "fp4 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "novita-vs-cloudflare",
      "a": "novita",
      "b": "cloudflare",
      "title": "Novita AI vs Cloudflare Workers AI",
      "seoTitle": "Novita AI vs Cloudflare Workers AI: price per model",
      "description": "Novita AI vs Cloudflare Workers AI: 11 list prices per million tokens for the same models, as reported by OpenRouter’s public API, with the gap stated.",
      "verdict": "Novita AI and Cloudflare Workers AI share 11 reported list prices from one study. No row names a winner: a reported list price has no interval, so each gap is stated, not ranked. Prices change often; the snapshot date is in each row’s context. The rows are 11 unclear; each row says why.",
      "rows": [
        {
          "metric": "Llama 3.3 70B Instruct: price per million tokens by provider (Input)",
          "aValue": 0.135,
          "bValue": 0.293,
          "unit": "usd",
          "aDisplay": "$0.14",
          "bDisplay": "$0.29",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.14 vs $0.29, 2.2x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-llama-3-3-70b-instruct",
          "aContext": "bf16 · Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Llama 3.3 70B Instruct: price per million tokens by provider (Output)",
          "aValue": 0.4,
          "bValue": 2.253,
          "unit": "usd",
          "aDisplay": "$0.40",
          "bDisplay": "$2.25",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.40 vs $2.25, 5.6x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-llama-3-3-70b-instruct",
          "aContext": "bf16 · Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "DeepSeek V4 Pro 0423: price per million tokens by provider (Input)",
          "aValue": 1.6,
          "bValue": 1.15,
          "unit": "usd",
          "aDisplay": "$1.60",
          "bDisplay": "$1.15",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($1.60 vs $1.15, 1.4x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-deepseek-v4-pro",
          "aContext": "fp8 · DeepSeek V4 Pro 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "DeepSeek V4 Pro 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "DeepSeek V4 Pro 0423: price per million tokens by provider (Output)",
          "aValue": 3.2,
          "bValue": 2.55,
          "unit": "usd",
          "aDisplay": "$3.20",
          "bDisplay": "$2.55",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($3.20 vs $2.55, 1.3x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-deepseek-v4-pro",
          "aContext": "fp8 · DeepSeek V4 Pro 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "DeepSeek V4 Pro 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "DeepSeek V4 Pro 0423: price per million tokens by provider (Cache read)",
          "aValue": 0.135,
          "bValue": 0.2,
          "unit": "usd",
          "aDisplay": "$0.14",
          "bDisplay": "$0.20",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.14 vs $0.20, 1.5x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-deepseek-v4-pro",
          "aContext": "fp8 · DeepSeek V4 Pro 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "DeepSeek V4 Pro 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "DeepSeek V4 Flash 0423: price per million tokens by provider (Input)",
          "aValue": 0.14,
          "bValue": 0.44,
          "unit": "usd",
          "aDisplay": "$0.14",
          "bDisplay": "$0.44",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.14 vs $0.44, 3.1x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-deepseek-v4-flash",
          "aContext": "fp8 · DeepSeek V4 Flash 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "DeepSeek V4 Flash 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "DeepSeek V4 Flash 0423: price per million tokens by provider (Output)",
          "aValue": 0.28,
          "bValue": 1.32,
          "unit": "usd",
          "aDisplay": "$0.28",
          "bDisplay": "$1.32",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.28 vs $1.32, 4.7x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-deepseek-v4-flash",
          "aContext": "fp8 · DeepSeek V4 Flash 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "DeepSeek V4 Flash 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "DeepSeek V4 Flash 0423: price per million tokens by provider (Cache read)",
          "aValue": 0.028,
          "bValue": 0.014,
          "unit": "usd",
          "aDisplay": "$0.028",
          "bDisplay": "$0.014",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.028 vs $0.014, 2.0x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-deepseek-v4-flash",
          "aContext": "fp8 · DeepSeek V4 Flash 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "DeepSeek V4 Flash 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "GLM 5.3: price per million tokens by provider (Input)",
          "aValue": 0.42,
          "bValue": 1.4,
          "unit": "usd",
          "aDisplay": "$0.42",
          "bDisplay": "$1.40",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.42 vs $1.40, 3.3x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "aContext": "fp8 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "GLM 5.3: price per million tokens by provider (Output)",
          "aValue": 1.32,
          "bValue": 4.4,
          "unit": "usd",
          "aDisplay": "$1.32",
          "bDisplay": "$4.40",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($1.32 vs $4.40, 3.3x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "aContext": "fp8 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "GLM 5.3: price per million tokens by provider (Cache read)",
          "aValue": 0.078,
          "bValue": 0.26,
          "unit": "usd",
          "aDisplay": "$0.078",
          "bDisplay": "$0.26",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.078 vs $0.26, 3.3x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "aContext": "fp8 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "novita-vs-siliconflow",
      "a": "novita",
      "b": "siliconflow",
      "title": "Novita AI vs SiliconFlow",
      "seoTitle": "Novita AI vs SiliconFlow: price per model",
      "description": "Novita AI vs SiliconFlow: 11 list prices per million tokens for the same models, as reported by OpenRouter’s public API, with the gap stated.",
      "verdict": "Novita AI and SiliconFlow share 11 reported list prices from one study. No row names a winner: a reported list price has no interval, so each gap is stated, not ranked. Prices change often; the snapshot date is in each row’s context. The rows are 3 ties and 8 unclear; each row says why.",
      "rows": [
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Input)",
          "aValue": 0.05,
          "bValue": 0.15,
          "unit": "usd",
          "aDisplay": "$0.050",
          "bDisplay": "$0.15",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.050 vs $0.15, 3.0x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Output)",
          "aValue": 0.25,
          "bValue": 0.6,
          "unit": "usd",
          "aDisplay": "$0.25",
          "bDisplay": "$0.60",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.25 vs $0.60, 2.4x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "DeepSeek V4 Pro 0423: price per million tokens by provider (Input)",
          "aValue": 1.6,
          "bValue": 1.50162,
          "unit": "usd",
          "aDisplay": "$1.60",
          "bDisplay": "$1.50",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($1.60 vs $1.50, 1.1x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-deepseek-v4-pro",
          "aContext": "fp8 · DeepSeek V4 Pro 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · DeepSeek V4 Pro 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "DeepSeek V4 Pro 0423: price per million tokens by provider (Output)",
          "aValue": 3.2,
          "bValue": 3.135,
          "unit": "usd",
          "aDisplay": "$3.20",
          "bDisplay": "$3.13",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($3.20 vs $3.13, 1.0x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-deepseek-v4-pro",
          "aContext": "fp8 · DeepSeek V4 Pro 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · DeepSeek V4 Pro 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "DeepSeek V4 Pro 0423: price per million tokens by provider (Cache read)",
          "aValue": 0.135,
          "bValue": 0.135,
          "unit": "usd",
          "aDisplay": "$0.14",
          "bDisplay": "$0.14",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-deepseek-v4-pro",
          "aContext": "fp8 · DeepSeek V4 Pro 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · DeepSeek V4 Pro 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "DeepSeek V4 Flash 0423: price per million tokens by provider (Input)",
          "aValue": 0.14,
          "bValue": 0.13,
          "unit": "usd",
          "aDisplay": "$0.14",
          "bDisplay": "$0.13",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.14 vs $0.13, 1.1x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-deepseek-v4-flash",
          "aContext": "fp8 · DeepSeek V4 Flash 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · DeepSeek V4 Flash 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "DeepSeek V4 Flash 0423: price per million tokens by provider (Output)",
          "aValue": 0.28,
          "bValue": 0.28,
          "unit": "usd",
          "aDisplay": "$0.28",
          "bDisplay": "$0.28",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-deepseek-v4-flash",
          "aContext": "fp8 · DeepSeek V4 Flash 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · DeepSeek V4 Flash 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "DeepSeek V4 Flash 0423: price per million tokens by provider (Cache read)",
          "aValue": 0.028,
          "bValue": 0.028,
          "unit": "usd",
          "aDisplay": "$0.028",
          "bDisplay": "$0.028",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-deepseek-v4-flash",
          "aContext": "fp8 · DeepSeek V4 Flash 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · DeepSeek V4 Flash 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "GLM 5.3: price per million tokens by provider (Input)",
          "aValue": 0.42,
          "bValue": 0.7,
          "unit": "usd",
          "aDisplay": "$0.42",
          "bDisplay": "$0.70",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.42 vs $0.70, 1.7x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "aContext": "fp8 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "GLM 5.3: price per million tokens by provider (Output)",
          "aValue": 1.32,
          "bValue": 2.2,
          "unit": "usd",
          "aDisplay": "$1.32",
          "bDisplay": "$2.20",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($1.32 vs $2.20, 1.7x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "aContext": "fp8 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "GLM 5.3: price per million tokens by provider (Cache read)",
          "aValue": 0.078,
          "bValue": 0.13,
          "unit": "usd",
          "aDisplay": "$0.078",
          "bDisplay": "$0.13",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.078 vs $0.13, 1.7x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "aContext": "fp8 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "parasail-vs-cloudflare",
      "a": "parasail",
      "b": "cloudflare",
      "title": "Parasail vs Cloudflare Workers AI",
      "seoTitle": "Parasail vs Cloudflare Workers AI: price per model",
      "description": "Parasail vs Cloudflare Workers AI: 11 list prices per million tokens for the same models, as reported by OpenRouter’s public API, with the gap stated.",
      "verdict": "Parasail and Cloudflare Workers AI share 11 reported list prices from one study. No row names a winner: a reported list price has no interval, so each gap is stated, not ranked. Prices change often; the snapshot date is in each row’s context. The rows are 3 ties and 8 unclear; each row says why.",
      "rows": [
        {
          "metric": "Llama 3.3 70B Instruct: price per million tokens by provider (Input)",
          "aValue": 0.22,
          "bValue": 0.293,
          "unit": "usd",
          "aDisplay": "$0.22",
          "bDisplay": "$0.29",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.22 vs $0.29, 1.3x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-llama-3-3-70b-instruct",
          "aContext": "fp8 · Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Llama 3.3 70B Instruct: price per million tokens by provider (Output)",
          "aValue": 0.5,
          "bValue": 2.253,
          "unit": "usd",
          "aDisplay": "$0.50",
          "bDisplay": "$2.25",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.50 vs $2.25, 4.5x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-llama-3-3-70b-instruct",
          "aContext": "fp8 · Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "DeepSeek V4 Pro 0423: price per million tokens by provider (Input)",
          "aValue": 0.45,
          "bValue": 1.15,
          "unit": "usd",
          "aDisplay": "$0.45",
          "bDisplay": "$1.15",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.45 vs $1.15, 2.6x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-deepseek-v4-pro",
          "aContext": "fp8 · DeepSeek V4 Pro 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "DeepSeek V4 Pro 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "DeepSeek V4 Pro 0423: price per million tokens by provider (Output)",
          "aValue": 3.48,
          "bValue": 2.55,
          "unit": "usd",
          "aDisplay": "$3.48",
          "bDisplay": "$2.55",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($3.48 vs $2.55, 1.4x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-deepseek-v4-pro",
          "aContext": "fp8 · DeepSeek V4 Pro 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "DeepSeek V4 Pro 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "DeepSeek V4 Pro 0423: price per million tokens by provider (Cache read)",
          "aValue": 0.1,
          "bValue": 0.2,
          "unit": "usd",
          "aDisplay": "$0.10",
          "bDisplay": "$0.20",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.10 vs $0.20, 2.0x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-deepseek-v4-pro",
          "aContext": "fp8 · DeepSeek V4 Pro 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "DeepSeek V4 Pro 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "DeepSeek V4 Flash 0423: price per million tokens by provider (Input)",
          "aValue": 0.14,
          "bValue": 0.44,
          "unit": "usd",
          "aDisplay": "$0.14",
          "bDisplay": "$0.44",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.14 vs $0.44, 3.1x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-deepseek-v4-flash",
          "aContext": "fp8 · DeepSeek V4 Flash 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "DeepSeek V4 Flash 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "DeepSeek V4 Flash 0423: price per million tokens by provider (Output)",
          "aValue": 0.28,
          "bValue": 1.32,
          "unit": "usd",
          "aDisplay": "$0.28",
          "bDisplay": "$1.32",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.28 vs $1.32, 4.7x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-deepseek-v4-flash",
          "aContext": "fp8 · DeepSeek V4 Flash 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "DeepSeek V4 Flash 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "DeepSeek V4 Flash 0423: price per million tokens by provider (Cache read)",
          "aValue": 0.07,
          "bValue": 0.014,
          "unit": "usd",
          "aDisplay": "$0.070",
          "bDisplay": "$0.014",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.070 vs $0.014, 5.0x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-deepseek-v4-flash",
          "aContext": "fp8 · DeepSeek V4 Flash 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "DeepSeek V4 Flash 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "GLM 5.3: price per million tokens by provider (Input)",
          "aValue": 1.4,
          "bValue": 1.4,
          "unit": "usd",
          "aDisplay": "$1.40",
          "bDisplay": "$1.40",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "aContext": "fp8 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "GLM 5.3: price per million tokens by provider (Output)",
          "aValue": 4.4,
          "bValue": 4.4,
          "unit": "usd",
          "aDisplay": "$4.40",
          "bDisplay": "$4.40",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "aContext": "fp8 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "GLM 5.3: price per million tokens by provider (Cache read)",
          "aValue": 0.26,
          "bValue": 0.26,
          "unit": "usd",
          "aDisplay": "$0.26",
          "bDisplay": "$0.26",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "aContext": "fp8 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "together-vs-deepinfra",
      "a": "together",
      "b": "deepinfra",
      "title": "Together AI vs DeepInfra",
      "seoTitle": "Together AI vs DeepInfra: price per model",
      "description": "Together AI vs DeepInfra: 10 list prices per million tokens for the same models, as reported by OpenRouter’s public API, with the gap stated.",
      "verdict": "Together AI and DeepInfra share 10 reported list prices from one study. No row names a winner: a reported list price has no interval, so each gap is stated, not ranked. Prices change often; the snapshot date is in each row’s context. The rows are 10 unclear; each row says why.",
      "rows": [
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Input)",
          "aValue": 0.15,
          "bValue": 0.037,
          "unit": "usd",
          "aDisplay": "$0.15",
          "bDisplay": "$0.037",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.15 vs $0.037, 4.1x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "bf16 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Output)",
          "aValue": 0.6,
          "bValue": 0.17,
          "unit": "usd",
          "aDisplay": "$0.60",
          "bDisplay": "$0.17",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.60 vs $0.17, 3.5x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "bf16 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Llama 3.3 70B Instruct: price per million tokens by provider (Input)",
          "aValue": 1.04,
          "bValue": 0.1,
          "unit": "usd",
          "aDisplay": "$1.04",
          "bDisplay": "$0.10",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($1.04 vs $0.10, 10x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-llama-3-3-70b-instruct",
          "aContext": "Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Llama 3.3 70B Instruct: price per million tokens by provider (Output)",
          "aValue": 1.04,
          "bValue": 0.32,
          "unit": "usd",
          "aDisplay": "$1.04",
          "bDisplay": "$0.32",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($1.04 vs $0.32, 3.3x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-llama-3-3-70b-instruct",
          "aContext": "Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Kimi K3: price per million tokens by provider (Input)",
          "aValue": 2.7,
          "bValue": 2.85,
          "unit": "usd",
          "aDisplay": "$2.70",
          "bDisplay": "$2.85",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($2.70 vs $2.85, 1.1x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-kimi-k3",
          "aContext": "Kimi K3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "mxfp4 · Kimi K3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Kimi K3: price per million tokens by provider (Output)",
          "aValue": 13.5,
          "bValue": 14.25,
          "unit": "usd",
          "aDisplay": "$13.50",
          "bDisplay": "$14.25",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($13.50 vs $14.25, 1.1x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-kimi-k3",
          "aContext": "Kimi K3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "mxfp4 · Kimi K3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Kimi K3: price per million tokens by provider (Cache read)",
          "aValue": 0.27,
          "bValue": 0.285,
          "unit": "usd",
          "aDisplay": "$0.27",
          "bDisplay": "$0.28",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.27 vs $0.28, 1.1x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-kimi-k3",
          "aContext": "Kimi K3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "mxfp4 · Kimi K3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "GLM 5.3: price per million tokens by provider (Input)",
          "aValue": 1.4,
          "bValue": 0.5625,
          "unit": "usd",
          "aDisplay": "$1.40",
          "bDisplay": "$0.56",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($1.40 vs $0.56, 2.5x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "aContext": "GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp4 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "GLM 5.3: price per million tokens by provider (Output)",
          "aValue": 4.4,
          "bValue": 2.5,
          "unit": "usd",
          "aDisplay": "$4.40",
          "bDisplay": "$2.50",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($4.40 vs $2.50, 1.8x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "aContext": "GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp4 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "GLM 5.3: price per million tokens by provider (Cache read)",
          "aValue": 0.26,
          "bValue": 0.125,
          "unit": "usd",
          "aDisplay": "$0.26",
          "bDisplay": "$0.13",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.26 vs $0.13, 2.1x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "aContext": "GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp4 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "together-vs-parasail",
      "a": "together",
      "b": "parasail",
      "title": "Together AI vs Parasail",
      "seoTitle": "Together AI vs Parasail: price per model",
      "description": "Together AI vs Parasail: 10 list prices per million tokens for the same models, as reported by OpenRouter’s public API, with the gap stated.",
      "verdict": "Together AI and Parasail share 10 reported list prices from one study. No row names a winner: a reported list price has no interval, so each gap is stated, not ranked. Prices change often; the snapshot date is in each row’s context. The rows are 3 ties and 7 unclear; each row says why.",
      "rows": [
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Input)",
          "aValue": 0.15,
          "bValue": 0.1,
          "unit": "usd",
          "aDisplay": "$0.15",
          "bDisplay": "$0.10",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.15 vs $0.10, 1.5x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Output)",
          "aValue": 0.6,
          "bValue": 0.75,
          "unit": "usd",
          "aDisplay": "$0.60",
          "bDisplay": "$0.75",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.60 vs $0.75, 1.3x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Llama 3.3 70B Instruct: price per million tokens by provider (Input)",
          "aValue": 1.04,
          "bValue": 0.22,
          "unit": "usd",
          "aDisplay": "$1.04",
          "bDisplay": "$0.22",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($1.04 vs $0.22, 4.7x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-llama-3-3-70b-instruct",
          "aContext": "Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Llama 3.3 70B Instruct: price per million tokens by provider (Output)",
          "aValue": 1.04,
          "bValue": 0.5,
          "unit": "usd",
          "aDisplay": "$1.04",
          "bDisplay": "$0.50",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($1.04 vs $0.50, 2.1x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-llama-3-3-70b-instruct",
          "aContext": "Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Kimi K3: price per million tokens by provider (Input)",
          "aValue": 2.7,
          "bValue": 3,
          "unit": "usd",
          "aDisplay": "$2.70",
          "bDisplay": "$3.00",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($2.70 vs $3.00, 1.1x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-kimi-k3",
          "aContext": "Kimi K3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp4 · Kimi K3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Kimi K3: price per million tokens by provider (Output)",
          "aValue": 13.5,
          "bValue": 15,
          "unit": "usd",
          "aDisplay": "$13.50",
          "bDisplay": "$15.00",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($13.50 vs $15.00, 1.1x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-kimi-k3",
          "aContext": "Kimi K3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp4 · Kimi K3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Kimi K3: price per million tokens by provider (Cache read)",
          "aValue": 0.27,
          "bValue": 0.3,
          "unit": "usd",
          "aDisplay": "$0.27",
          "bDisplay": "$0.30",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.27 vs $0.30, 1.1x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-kimi-k3",
          "aContext": "Kimi K3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp4 · Kimi K3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "GLM 5.3: price per million tokens by provider (Input)",
          "aValue": 1.4,
          "bValue": 1.4,
          "unit": "usd",
          "aDisplay": "$1.40",
          "bDisplay": "$1.40",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "aContext": "GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "GLM 5.3: price per million tokens by provider (Output)",
          "aValue": 4.4,
          "bValue": 4.4,
          "unit": "usd",
          "aDisplay": "$4.40",
          "bDisplay": "$4.40",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "aContext": "GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "GLM 5.3: price per million tokens by provider (Cache read)",
          "aValue": 0.26,
          "bValue": 0.26,
          "unit": "usd",
          "aDisplay": "$0.26",
          "bDisplay": "$0.26",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "aContext": "GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "cloudflare-vs-siliconflow",
      "a": "cloudflare",
      "b": "siliconflow",
      "title": "Cloudflare Workers AI vs SiliconFlow",
      "seoTitle": "Cloudflare Workers AI vs SiliconFlow: price per model",
      "description": "Cloudflare Workers AI vs SiliconFlow: 9 list prices per million tokens for the same models, as reported by OpenRouter’s public API, with the gap stated.",
      "verdict": "Cloudflare Workers AI and SiliconFlow share 9 reported list prices from one study. No row names a winner: a reported list price has no interval, so each gap is stated, not ranked. Prices change often; the snapshot date is in each row’s context. The rows are 9 unclear; each row says why.",
      "rows": [
        {
          "metric": "DeepSeek V4 Pro 0423: price per million tokens by provider (Input)",
          "aValue": 1.15,
          "bValue": 1.50162,
          "unit": "usd",
          "aDisplay": "$1.15",
          "bDisplay": "$1.50",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($1.15 vs $1.50, 1.3x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-deepseek-v4-pro",
          "aContext": "DeepSeek V4 Pro 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · DeepSeek V4 Pro 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "DeepSeek V4 Pro 0423: price per million tokens by provider (Output)",
          "aValue": 2.55,
          "bValue": 3.135,
          "unit": "usd",
          "aDisplay": "$2.55",
          "bDisplay": "$3.13",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($2.55 vs $3.13, 1.2x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-deepseek-v4-pro",
          "aContext": "DeepSeek V4 Pro 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · DeepSeek V4 Pro 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "DeepSeek V4 Pro 0423: price per million tokens by provider (Cache read)",
          "aValue": 0.2,
          "bValue": 0.135,
          "unit": "usd",
          "aDisplay": "$0.20",
          "bDisplay": "$0.14",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.20 vs $0.14, 1.5x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-deepseek-v4-pro",
          "aContext": "DeepSeek V4 Pro 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · DeepSeek V4 Pro 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "DeepSeek V4 Flash 0423: price per million tokens by provider (Input)",
          "aValue": 0.44,
          "bValue": 0.13,
          "unit": "usd",
          "aDisplay": "$0.44",
          "bDisplay": "$0.13",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.44 vs $0.13, 3.4x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-deepseek-v4-flash",
          "aContext": "DeepSeek V4 Flash 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · DeepSeek V4 Flash 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "DeepSeek V4 Flash 0423: price per million tokens by provider (Output)",
          "aValue": 1.32,
          "bValue": 0.28,
          "unit": "usd",
          "aDisplay": "$1.32",
          "bDisplay": "$0.28",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($1.32 vs $0.28, 4.7x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-deepseek-v4-flash",
          "aContext": "DeepSeek V4 Flash 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · DeepSeek V4 Flash 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "DeepSeek V4 Flash 0423: price per million tokens by provider (Cache read)",
          "aValue": 0.014,
          "bValue": 0.028,
          "unit": "usd",
          "aDisplay": "$0.014",
          "bDisplay": "$0.028",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.014 vs $0.028, 2.0x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-deepseek-v4-flash",
          "aContext": "DeepSeek V4 Flash 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · DeepSeek V4 Flash 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "GLM 5.3: price per million tokens by provider (Input)",
          "aValue": 1.4,
          "bValue": 0.7,
          "unit": "usd",
          "aDisplay": "$1.40",
          "bDisplay": "$0.70",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($1.40 vs $0.70, 2.0x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "aContext": "GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "GLM 5.3: price per million tokens by provider (Output)",
          "aValue": 4.4,
          "bValue": 2.2,
          "unit": "usd",
          "aDisplay": "$4.40",
          "bDisplay": "$2.20",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($4.40 vs $2.20, 2.0x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "aContext": "GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "GLM 5.3: price per million tokens by provider (Cache read)",
          "aValue": 0.26,
          "bValue": 0.13,
          "unit": "usd",
          "aDisplay": "$0.26",
          "bDisplay": "$0.13",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.26 vs $0.13, 2.0x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "aContext": "GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "parasail-vs-baseten",
      "a": "parasail",
      "b": "baseten",
      "title": "Parasail vs Baseten",
      "seoTitle": "Parasail vs Baseten: price per model",
      "description": "Parasail vs Baseten: 9 list prices per million tokens for the same models, as reported by OpenRouter’s public API, with the gap stated.",
      "verdict": "Parasail and Baseten share 9 reported list prices from one study. No row names a winner: a reported list price has no interval, so each gap is stated, not ranked. Prices change often; the snapshot date is in each row’s context. The rows are 6 ties and 3 unclear; each row says why.",
      "rows": [
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Input)",
          "aValue": 0.1,
          "bValue": 0.1,
          "unit": "usd",
          "aDisplay": "$0.10",
          "bDisplay": "$0.10",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Output)",
          "aValue": 0.75,
          "bValue": 0.5,
          "unit": "usd",
          "aDisplay": "$0.75",
          "bDisplay": "$0.50",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.75 vs $0.50, 1.5x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Cache read)",
          "aValue": 0.055,
          "bValue": 0.1,
          "unit": "usd",
          "aDisplay": "$0.055",
          "bDisplay": "$0.10",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.055 vs $0.10, 1.8x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Kimi K3: price per million tokens by provider (Input)",
          "aValue": 3,
          "bValue": 3,
          "unit": "usd",
          "aDisplay": "$3.00",
          "bDisplay": "$3.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-kimi-k3",
          "aContext": "fp4 · Kimi K3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · Kimi K3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Kimi K3: price per million tokens by provider (Output)",
          "aValue": 15,
          "bValue": 15,
          "unit": "usd",
          "aDisplay": "$15.00",
          "bDisplay": "$15.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-kimi-k3",
          "aContext": "fp4 · Kimi K3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · Kimi K3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Kimi K3: price per million tokens by provider (Cache read)",
          "aValue": 0.3,
          "bValue": 0.3,
          "unit": "usd",
          "aDisplay": "$0.30",
          "bDisplay": "$0.30",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-kimi-k3",
          "aContext": "fp4 · Kimi K3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · Kimi K3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "GLM 5.3: price per million tokens by provider (Input)",
          "aValue": 1.4,
          "bValue": 1.4,
          "unit": "usd",
          "aDisplay": "$1.40",
          "bDisplay": "$1.40",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "aContext": "fp8 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp4 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "GLM 5.3: price per million tokens by provider (Output)",
          "aValue": 4.4,
          "bValue": 4.4,
          "unit": "usd",
          "aDisplay": "$4.40",
          "bDisplay": "$4.40",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "aContext": "fp8 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp4 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "GLM 5.3: price per million tokens by provider (Cache read)",
          "aValue": 0.26,
          "bValue": 0.14,
          "unit": "usd",
          "aDisplay": "$0.26",
          "bDisplay": "$0.14",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.26 vs $0.14, 1.9x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "aContext": "fp8 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp4 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "deepinfra-vs-baseten",
      "a": "deepinfra",
      "b": "baseten",
      "title": "DeepInfra vs Baseten",
      "seoTitle": "DeepInfra vs Baseten: price per model",
      "description": "DeepInfra vs Baseten: 8 list prices per million tokens for the same models, as reported by OpenRouter’s public API, with the gap stated.",
      "verdict": "DeepInfra and Baseten share 8 reported list prices from one study. No row names a winner: a reported list price has no interval, so each gap is stated, not ranked. Prices change often; the snapshot date is in each row’s context. The rows are 8 unclear; each row says why.",
      "rows": [
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Input)",
          "aValue": 0.037,
          "bValue": 0.1,
          "unit": "usd",
          "aDisplay": "$0.037",
          "bDisplay": "$0.10",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.037 vs $0.10, 2.7x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "bf16 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Output)",
          "aValue": 0.17,
          "bValue": 0.5,
          "unit": "usd",
          "aDisplay": "$0.17",
          "bDisplay": "$0.50",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.17 vs $0.50, 2.9x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "bf16 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Kimi K3: price per million tokens by provider (Input)",
          "aValue": 2.85,
          "bValue": 3,
          "unit": "usd",
          "aDisplay": "$2.85",
          "bDisplay": "$3.00",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($2.85 vs $3.00, 1.1x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-kimi-k3",
          "aContext": "mxfp4 · Kimi K3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · Kimi K3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Kimi K3: price per million tokens by provider (Output)",
          "aValue": 14.25,
          "bValue": 15,
          "unit": "usd",
          "aDisplay": "$14.25",
          "bDisplay": "$15.00",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($14.25 vs $15.00, 1.1x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-kimi-k3",
          "aContext": "mxfp4 · Kimi K3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · Kimi K3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Kimi K3: price per million tokens by provider (Cache read)",
          "aValue": 0.285,
          "bValue": 0.3,
          "unit": "usd",
          "aDisplay": "$0.28",
          "bDisplay": "$0.30",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.28 vs $0.30, 1.1x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-kimi-k3",
          "aContext": "mxfp4 · Kimi K3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · Kimi K3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "GLM 5.3: price per million tokens by provider (Input)",
          "aValue": 0.5625,
          "bValue": 1.4,
          "unit": "usd",
          "aDisplay": "$0.56",
          "bDisplay": "$1.40",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.56 vs $1.40, 2.5x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "aContext": "fp4 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp4 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "GLM 5.3: price per million tokens by provider (Output)",
          "aValue": 2.5,
          "bValue": 4.4,
          "unit": "usd",
          "aDisplay": "$2.50",
          "bDisplay": "$4.40",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($2.50 vs $4.40, 1.8x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "aContext": "fp4 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp4 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "GLM 5.3: price per million tokens by provider (Cache read)",
          "aValue": 0.125,
          "bValue": 0.14,
          "unit": "usd",
          "aDisplay": "$0.13",
          "bDisplay": "$0.14",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.13 vs $0.14, 1.1x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "aContext": "fp4 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp4 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "together-vs-baseten",
      "a": "together",
      "b": "baseten",
      "title": "Together AI vs Baseten",
      "seoTitle": "Together AI vs Baseten: price per model",
      "description": "Together AI vs Baseten: 8 list prices per million tokens for the same models, as reported by OpenRouter’s public API, with the gap stated.",
      "verdict": "Together AI and Baseten share 8 reported list prices from one study. No row names a winner: a reported list price has no interval, so each gap is stated, not ranked. Prices change often; the snapshot date is in each row’s context. The rows are 2 ties and 6 unclear; each row says why.",
      "rows": [
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Input)",
          "aValue": 0.15,
          "bValue": 0.1,
          "unit": "usd",
          "aDisplay": "$0.15",
          "bDisplay": "$0.10",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.15 vs $0.10, 1.5x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Output)",
          "aValue": 0.6,
          "bValue": 0.5,
          "unit": "usd",
          "aDisplay": "$0.60",
          "bDisplay": "$0.50",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.60 vs $0.50, 1.2x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Kimi K3: price per million tokens by provider (Input)",
          "aValue": 2.7,
          "bValue": 3,
          "unit": "usd",
          "aDisplay": "$2.70",
          "bDisplay": "$3.00",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($2.70 vs $3.00, 1.1x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-kimi-k3",
          "aContext": "Kimi K3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · Kimi K3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Kimi K3: price per million tokens by provider (Output)",
          "aValue": 13.5,
          "bValue": 15,
          "unit": "usd",
          "aDisplay": "$13.50",
          "bDisplay": "$15.00",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($13.50 vs $15.00, 1.1x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-kimi-k3",
          "aContext": "Kimi K3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · Kimi K3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Kimi K3: price per million tokens by provider (Cache read)",
          "aValue": 0.27,
          "bValue": 0.3,
          "unit": "usd",
          "aDisplay": "$0.27",
          "bDisplay": "$0.30",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.27 vs $0.30, 1.1x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-kimi-k3",
          "aContext": "Kimi K3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · Kimi K3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "GLM 5.3: price per million tokens by provider (Input)",
          "aValue": 1.4,
          "bValue": 1.4,
          "unit": "usd",
          "aDisplay": "$1.40",
          "bDisplay": "$1.40",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "aContext": "GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp4 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "GLM 5.3: price per million tokens by provider (Output)",
          "aValue": 4.4,
          "bValue": 4.4,
          "unit": "usd",
          "aDisplay": "$4.40",
          "bDisplay": "$4.40",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "aContext": "GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp4 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "GLM 5.3: price per million tokens by provider (Cache read)",
          "aValue": 0.26,
          "bValue": 0.14,
          "unit": "usd",
          "aDisplay": "$0.26",
          "bDisplay": "$0.14",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.26 vs $0.14, 1.9x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "aContext": "GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp4 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "together-vs-novita",
      "a": "together",
      "b": "novita",
      "title": "Together AI vs Novita AI",
      "seoTitle": "Together AI vs Novita AI: price per model",
      "description": "Together AI vs Novita AI: 7 list prices per million tokens for the same models, as reported by OpenRouter’s public API, with the gap stated.",
      "verdict": "Together AI and Novita AI share 7 reported list prices from one study. No row names a winner: a reported list price has no interval, so each gap is stated, not ranked. Prices change often; the snapshot date is in each row’s context. The rows are 7 unclear; each row says why.",
      "rows": [
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Input)",
          "aValue": 0.15,
          "bValue": 0.05,
          "unit": "usd",
          "aDisplay": "$0.15",
          "bDisplay": "$0.050",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.15 vs $0.050, 3.0x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Output)",
          "aValue": 0.6,
          "bValue": 0.25,
          "unit": "usd",
          "aDisplay": "$0.60",
          "bDisplay": "$0.25",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.60 vs $0.25, 2.4x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Llama 3.3 70B Instruct: price per million tokens by provider (Input)",
          "aValue": 1.04,
          "bValue": 0.135,
          "unit": "usd",
          "aDisplay": "$1.04",
          "bDisplay": "$0.14",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($1.04 vs $0.14, 7.7x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-llama-3-3-70b-instruct",
          "aContext": "Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "bf16 · Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Llama 3.3 70B Instruct: price per million tokens by provider (Output)",
          "aValue": 1.04,
          "bValue": 0.4,
          "unit": "usd",
          "aDisplay": "$1.04",
          "bDisplay": "$0.40",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($1.04 vs $0.40, 2.6x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-llama-3-3-70b-instruct",
          "aContext": "Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "bf16 · Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "GLM 5.3: price per million tokens by provider (Input)",
          "aValue": 1.4,
          "bValue": 0.42,
          "unit": "usd",
          "aDisplay": "$1.40",
          "bDisplay": "$0.42",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($1.40 vs $0.42, 3.3x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "aContext": "GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "GLM 5.3: price per million tokens by provider (Output)",
          "aValue": 4.4,
          "bValue": 1.32,
          "unit": "usd",
          "aDisplay": "$4.40",
          "bDisplay": "$1.32",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($4.40 vs $1.32, 3.3x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "aContext": "GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "GLM 5.3: price per million tokens by provider (Cache read)",
          "aValue": 0.26,
          "bValue": 0.078,
          "unit": "usd",
          "aDisplay": "$0.26",
          "bDisplay": "$0.078",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.26 vs $0.078, 3.3x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "aContext": "GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "baseten-vs-siliconflow",
      "a": "baseten",
      "b": "siliconflow",
      "title": "Baseten vs SiliconFlow",
      "seoTitle": "Baseten vs SiliconFlow: price per model",
      "description": "Baseten vs SiliconFlow: 6 list prices per million tokens for the same models, as reported by OpenRouter’s public API, with the gap stated.",
      "verdict": "Baseten and SiliconFlow share 6 reported list prices from one study. No row names a winner: a reported list price has no interval, so each gap is stated, not ranked. Prices change often; the snapshot date is in each row’s context. The rows are 6 unclear; each row says why.",
      "rows": [
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Input)",
          "aValue": 0.1,
          "bValue": 0.15,
          "unit": "usd",
          "aDisplay": "$0.10",
          "bDisplay": "$0.15",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.10 vs $0.15, 1.5x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Output)",
          "aValue": 0.5,
          "bValue": 0.6,
          "unit": "usd",
          "aDisplay": "$0.50",
          "bDisplay": "$0.60",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.50 vs $0.60, 1.2x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Cache read)",
          "aValue": 0.1,
          "bValue": 0.075,
          "unit": "usd",
          "aDisplay": "$0.10",
          "bDisplay": "$0.075",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.10 vs $0.075, 1.3x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "GLM 5.3: price per million tokens by provider (Input)",
          "aValue": 1.4,
          "bValue": 0.7,
          "unit": "usd",
          "aDisplay": "$1.40",
          "bDisplay": "$0.70",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($1.40 vs $0.70, 2.0x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "aContext": "fp4 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "GLM 5.3: price per million tokens by provider (Output)",
          "aValue": 4.4,
          "bValue": 2.2,
          "unit": "usd",
          "aDisplay": "$4.40",
          "bDisplay": "$2.20",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($4.40 vs $2.20, 2.0x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "aContext": "fp4 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "GLM 5.3: price per million tokens by provider (Cache read)",
          "aValue": 0.14,
          "bValue": 0.13,
          "unit": "usd",
          "aDisplay": "$0.14",
          "bDisplay": "$0.13",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.14 vs $0.13, 1.1x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "aContext": "fp4 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "fireworks-vs-baseten",
      "a": "fireworks",
      "b": "baseten",
      "title": "Fireworks AI vs Baseten",
      "seoTitle": "Fireworks AI vs Baseten: price per model",
      "description": "Fireworks AI vs Baseten: 6 list prices per million tokens for the same models, as reported by OpenRouter’s public API, with the gap stated.",
      "verdict": "Fireworks AI and Baseten share 6 reported list prices from one study. No row names a winner: a reported list price has no interval, so each gap is stated, not ranked. Prices change often; the snapshot date is in each row’s context. The rows are 5 ties and 1 unclear; each row says why.",
      "rows": [
        {
          "metric": "Kimi K3: price per million tokens by provider (Input)",
          "aValue": 3,
          "bValue": 3,
          "unit": "usd",
          "aDisplay": "$3.00",
          "bDisplay": "$3.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-kimi-k3",
          "aContext": "Kimi K3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · Kimi K3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Kimi K3: price per million tokens by provider (Output)",
          "aValue": 15,
          "bValue": 15,
          "unit": "usd",
          "aDisplay": "$15.00",
          "bDisplay": "$15.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-kimi-k3",
          "aContext": "Kimi K3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · Kimi K3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Kimi K3: price per million tokens by provider (Cache read)",
          "aValue": 0.3,
          "bValue": 0.3,
          "unit": "usd",
          "aDisplay": "$0.30",
          "bDisplay": "$0.30",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-kimi-k3",
          "aContext": "Kimi K3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · Kimi K3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "GLM 5.3: price per million tokens by provider (Input)",
          "aValue": 1.4,
          "bValue": 1.4,
          "unit": "usd",
          "aDisplay": "$1.40",
          "bDisplay": "$1.40",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "aContext": "GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp4 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "GLM 5.3: price per million tokens by provider (Output)",
          "aValue": 4.4,
          "bValue": 4.4,
          "unit": "usd",
          "aDisplay": "$4.40",
          "bDisplay": "$4.40",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "aContext": "GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp4 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "GLM 5.3: price per million tokens by provider (Cache read)",
          "aValue": 0.26,
          "bValue": 0.14,
          "unit": "usd",
          "aDisplay": "$0.26",
          "bDisplay": "$0.14",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.26 vs $0.14, 1.9x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "aContext": "GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp4 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "fireworks-vs-deepinfra",
      "a": "fireworks",
      "b": "deepinfra",
      "title": "Fireworks AI vs DeepInfra",
      "seoTitle": "Fireworks AI vs DeepInfra: price per model",
      "description": "Fireworks AI vs DeepInfra: 6 list prices per million tokens for the same models, as reported by OpenRouter’s public API, with the gap stated.",
      "verdict": "Fireworks AI and DeepInfra share 6 reported list prices from one study. No row names a winner: a reported list price has no interval, so each gap is stated, not ranked. Prices change often; the snapshot date is in each row’s context. The rows are 6 unclear; each row says why.",
      "rows": [
        {
          "metric": "Kimi K3: price per million tokens by provider (Input)",
          "aValue": 3,
          "bValue": 2.85,
          "unit": "usd",
          "aDisplay": "$3.00",
          "bDisplay": "$2.85",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($3.00 vs $2.85, 1.1x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-kimi-k3",
          "aContext": "Kimi K3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "mxfp4 · Kimi K3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Kimi K3: price per million tokens by provider (Output)",
          "aValue": 15,
          "bValue": 14.25,
          "unit": "usd",
          "aDisplay": "$15.00",
          "bDisplay": "$14.25",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($15.00 vs $14.25, 1.1x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-kimi-k3",
          "aContext": "Kimi K3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "mxfp4 · Kimi K3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Kimi K3: price per million tokens by provider (Cache read)",
          "aValue": 0.3,
          "bValue": 0.285,
          "unit": "usd",
          "aDisplay": "$0.30",
          "bDisplay": "$0.28",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.30 vs $0.28, 1.1x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-kimi-k3",
          "aContext": "Kimi K3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "mxfp4 · Kimi K3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "GLM 5.3: price per million tokens by provider (Input)",
          "aValue": 1.4,
          "bValue": 0.5625,
          "unit": "usd",
          "aDisplay": "$1.40",
          "bDisplay": "$0.56",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($1.40 vs $0.56, 2.5x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "aContext": "GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp4 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "GLM 5.3: price per million tokens by provider (Output)",
          "aValue": 4.4,
          "bValue": 2.5,
          "unit": "usd",
          "aDisplay": "$4.40",
          "bDisplay": "$2.50",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($4.40 vs $2.50, 1.8x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "aContext": "GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp4 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "GLM 5.3: price per million tokens by provider (Cache read)",
          "aValue": 0.26,
          "bValue": 0.125,
          "unit": "usd",
          "aDisplay": "$0.26",
          "bDisplay": "$0.13",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.26 vs $0.13, 2.1x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "aContext": "GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp4 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "fireworks-vs-parasail",
      "a": "fireworks",
      "b": "parasail",
      "title": "Fireworks AI vs Parasail",
      "seoTitle": "Fireworks AI vs Parasail: price per model",
      "description": "Fireworks AI vs Parasail: 6 list prices per million tokens for the same models, as reported by OpenRouter’s public API, with the gap stated.",
      "verdict": "Fireworks AI and Parasail share 6 reported list prices from one study. No row names a winner: a reported list price has no interval, so each gap is stated, not ranked. Prices change often; the snapshot date is in each row’s context. The rows are 6 ties; each row says why.",
      "rows": [
        {
          "metric": "Kimi K3: price per million tokens by provider (Input)",
          "aValue": 3,
          "bValue": 3,
          "unit": "usd",
          "aDisplay": "$3.00",
          "bDisplay": "$3.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-kimi-k3",
          "aContext": "Kimi K3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp4 · Kimi K3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Kimi K3: price per million tokens by provider (Output)",
          "aValue": 15,
          "bValue": 15,
          "unit": "usd",
          "aDisplay": "$15.00",
          "bDisplay": "$15.00",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-kimi-k3",
          "aContext": "Kimi K3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp4 · Kimi K3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Kimi K3: price per million tokens by provider (Cache read)",
          "aValue": 0.3,
          "bValue": 0.3,
          "unit": "usd",
          "aDisplay": "$0.30",
          "bDisplay": "$0.30",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-kimi-k3",
          "aContext": "Kimi K3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp4 · Kimi K3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "GLM 5.3: price per million tokens by provider (Input)",
          "aValue": 1.4,
          "bValue": 1.4,
          "unit": "usd",
          "aDisplay": "$1.40",
          "bDisplay": "$1.40",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "aContext": "GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "GLM 5.3: price per million tokens by provider (Output)",
          "aValue": 4.4,
          "bValue": 4.4,
          "unit": "usd",
          "aDisplay": "$4.40",
          "bDisplay": "$4.40",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "aContext": "GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "GLM 5.3: price per million tokens by provider (Cache read)",
          "aValue": 0.26,
          "bValue": 0.26,
          "unit": "usd",
          "aDisplay": "$0.26",
          "bDisplay": "$0.26",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "aContext": "GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "gpt-6-1-sol-codex-cli-vs-gpt-6-luna-codex-cli",
      "a": "gpt-6-1-sol-codex-cli",
      "b": "gpt-6-luna-codex-cli",
      "title": "GPT-6.1 Sol (Codex CLI) vs GPT-6 Luna (Codex CLI)",
      "seoTitle": "GPT-6.1 Sol (Codex CLI) vs GPT-6 Luna (Codex CLI)",
      "description": "GPT-6.1 Sol (Codex CLI) vs GPT-6 Luna (Codex CLI): 6 measured metrics from 2 studies, with sample sizes, intervals and every failure counted.",
      "verdict": "GPT-6.1 Sol (Codex CLI) and GPT-6 Luna (Codex CLI) share 6 measured metrics and 2 list-price calculations from 2 studies. No row separates them: every interval or run range overlaps, too few runs were recorded, no interval was recorded, or more is not better for that metric. The rows are 8 unclear; each row says why. Calculation rows are derived from list prices and recorded counts; they are not bills or runs. Some rows rest on small samples (n = 3 at the smallest).",
      "rows": [
        {
          "metric": "CLI vs API: time for a one-line answer (Total time)",
          "aValue": 4.19,
          "bValue": 3.19,
          "unit": "seconds",
          "aDisplay": "4.19 s",
          "bDisplay": "3.19 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (GPT-6.1 Sol (Codex CLI) 3.81 s to 4.69 s; GPT-6 Luna (Codex CLI) 2.88 s to 3.83 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "cli-model-latency-tokens",
          "n": 5,
          "chartId": "cli-vs-api-exact-reply-latency",
          "aN": 5,
          "bN": 5,
          "aContext": "Codex CLI · effort high · fixed exact reply, 5 runs",
          "bContext": "Codex CLI · effort none · fixed exact reply, 5 runs",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            3.81,
            4.69
          ],
          "bRange": [
            2.88,
            3.83
          ]
        },
        {
          "metric": "CLI vs API: time for a one-line answer (First useful output)",
          "aValue": 3.79,
          "bValue": 2.79,
          "unit": "seconds",
          "aDisplay": "3.79 s",
          "bDisplay": "2.79 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (GPT-6.1 Sol (Codex CLI) 3.37 s to 4.30 s; GPT-6 Luna (Codex CLI) 2.46 s to 3.42 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "cli-model-latency-tokens",
          "n": 5,
          "chartId": "cli-vs-api-exact-reply-latency",
          "aN": 5,
          "bN": 5,
          "aContext": "Codex CLI · effort high · fixed exact reply, 5 runs",
          "bContext": "Codex CLI · effort none · fixed exact reply, 5 runs",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            3.37,
            4.3
          ],
          "bRange": [
            2.46,
            3.42
          ]
        },
        {
          "metric": "CLI vs API: time for a small coding task (Total time)",
          "aValue": 17.85,
          "bValue": 9.23,
          "unit": "seconds",
          "aDisplay": "17.9 s",
          "bDisplay": "9.23 s",
          "winner": "unclear",
          "basis": "Only 3 runs per side; the run ranges (fastest to slowest) do not overlap (GPT-6.1 Sol (Codex CLI) 17.7 s to 22.4 s; GPT-6 Luna (Codex CLI) 8.99 s to 11.7 s), but 3 runs cannot show a reliable difference. A range is not a confidence interval.",
          "studySlug": "cli-model-latency-tokens",
          "n": 3,
          "chartId": "cli-vs-api-small-coding-latency",
          "aN": 3,
          "bN": 3,
          "aContext": "Codex CLI · effort high · small coding task, 3 runs",
          "bContext": "Codex CLI · effort none · small coding task, 3 runs",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            17.68,
            22.42
          ],
          "bRange": [
            8.99,
            11.68
          ]
        },
        {
          "metric": "CLI vs API: time for a small coding task (First useful output)",
          "aValue": 17.27,
          "bValue": 8.68,
          "unit": "seconds",
          "aDisplay": "17.3 s",
          "bDisplay": "8.68 s",
          "winner": "unclear",
          "basis": "Only 3 runs per side; the run ranges (fastest to slowest) do not overlap (GPT-6.1 Sol (Codex CLI) 17.1 s to 21.9 s; GPT-6 Luna (Codex CLI) 8.27 s to 11.0 s), but 3 runs cannot show a reliable difference. A range is not a confidence interval.",
          "studySlug": "cli-model-latency-tokens",
          "n": 3,
          "chartId": "cli-vs-api-small-coding-latency",
          "aN": 3,
          "bN": 3,
          "aContext": "Codex CLI · effort high · small coding task, 3 runs",
          "bContext": "Codex CLI · effort none · small coding task, 3 runs",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            17.13,
            21.86
          ],
          "bRange": [
            8.27,
            11.01
          ]
        },
        {
          "metric": "Hidden prompt: input tokens for the same one-line request",
          "aValue": 19555,
          "bValue": 18859,
          "unit": "tokens",
          "aDisplay": "19,555",
          "bDisplay": "18,859",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "cli-model-latency-tokens",
          "n": 5,
          "chartId": "cli-vs-api-prompt-overhead",
          "aN": 5,
          "bN": 5,
          "aContext": "Codex CLI · effort high · short fixed tasks",
          "bContext": "Codex CLI · effort none · short fixed tasks"
        },
        {
          "metric": "Time to first text: a 250-line answer, six models",
          "aValue": 3.52,
          "bValue": 3.3,
          "unit": "seconds",
          "aDisplay": "3.52 s",
          "bDisplay": "3.30 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (GPT-6.1 Sol (Codex CLI) 2.75 s to 4.42 s; GPT-6 Luna (Codex CLI) 3.19 s to 3.47 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "llm-speed-anatomy",
          "n": 4,
          "chartId": "speed-anatomy-first-text",
          "aN": 4,
          "bN": 4,
          "aContext": "Codex CLI · effort low",
          "bContext": "Codex CLI · effort low",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            2.75,
            4.42
          ],
          "bRange": [
            3.19,
            3.47
          ]
        },
        {
          "metric": "Output speed after the first text: visible tokens per second (calculation)",
          "aValue": 79.6,
          "bValue": 129.1,
          "unit": "tokens",
          "aDisplay": "80",
          "bDisplay": "129",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "llm-speed-anatomy",
          "n": 4,
          "chartId": "speed-anatomy-output-speed",
          "aN": 4,
          "bN": 4,
          "aContext": "Codex CLI · effort low",
          "bContext": "Codex CLI · effort low",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            71.6,
            80.5
          ],
          "bRange": [
            55.5,
            259.1
          ],
          "calculation": true
        },
        {
          "metric": "Output speed in characters per second after the first text (calculation)",
          "aValue": 323,
          "bValue": 524,
          "unit": "count",
          "aDisplay": "323",
          "bDisplay": "524",
          "winner": "unclear",
          "basis": "More or fewer count is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "llm-speed-anatomy",
          "n": 4,
          "chartId": "speed-anatomy-chars-per-second",
          "aN": 4,
          "bN": 4,
          "aContext": "Codex CLI · effort low",
          "bContext": "Codex CLI · effort low",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            291,
            327
          ],
          "bRange": [
            225,
            1052
          ],
          "calculation": true
        }
      ]
    },
    {
      "slug": "groq-vs-parasail",
      "a": "groq",
      "b": "parasail",
      "title": "Groq vs Parasail",
      "seoTitle": "Groq vs Parasail: price per model",
      "description": "Groq vs Parasail: 6 list prices per million tokens for the same models, as reported by OpenRouter’s public API, with the gap stated.",
      "verdict": "Groq and Parasail share 6 reported list prices from one study. No row names a winner: a reported list price has no interval, so each gap is stated, not ranked. Prices change often; the snapshot date is in each row’s context. The rows are 6 unclear; each row says why.",
      "rows": [
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Input)",
          "aValue": 0.15,
          "bValue": 0.1,
          "unit": "usd",
          "aDisplay": "$0.15",
          "bDisplay": "$0.10",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.15 vs $0.10, 1.5x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Output)",
          "aValue": 0.6,
          "bValue": 0.75,
          "unit": "usd",
          "aDisplay": "$0.60",
          "bDisplay": "$0.75",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.60 vs $0.75, 1.3x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Cache read)",
          "aValue": 0.075,
          "bValue": 0.055,
          "unit": "usd",
          "aDisplay": "$0.075",
          "bDisplay": "$0.055",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.075 vs $0.055, 1.4x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Llama 3.3 70B Instruct: price per million tokens by provider (Input)",
          "aValue": 0.59,
          "bValue": 0.22,
          "unit": "usd",
          "aDisplay": "$0.59",
          "bDisplay": "$0.22",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.59 vs $0.22, 2.7x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-llama-3-3-70b-instruct",
          "aContext": "Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Llama 3.3 70B Instruct: price per million tokens by provider (Output)",
          "aValue": 0.79,
          "bValue": 0.5,
          "unit": "usd",
          "aDisplay": "$0.79",
          "bDisplay": "$0.50",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.79 vs $0.50, 1.6x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-llama-3-3-70b-instruct",
          "aContext": "Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Llama 3.3 70B Instruct: price per million tokens by provider (Cache read)",
          "aValue": 0.295,
          "bValue": 0.11,
          "unit": "usd",
          "aDisplay": "$0.29",
          "bDisplay": "$0.11",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.29 vs $0.11, 2.7x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-llama-3-3-70b-instruct",
          "aContext": "Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "gpt-6-1-sol-codex-cli-vs-gpt-6-luna-openai-api",
      "a": "gpt-6-1-sol-codex-cli",
      "b": "gpt-6-luna-openai-api",
      "title": "GPT-6.1 Sol (Codex CLI) vs GPT-6 Luna (OpenAI API)",
      "seoTitle": "GPT-6.1 Sol (Codex CLI) vs GPT-6 Luna (OpenAI API)",
      "description": "GPT-6.1 Sol (Codex CLI) vs GPT-6 Luna (OpenAI API): 5 measured metrics from one study, with sample sizes, intervals and every failure counted.",
      "verdict": "GPT-6.1 Sol (Codex CLI) and GPT-6 Luna (OpenAI API) share 5 measured metrics from one study. GPT-6 Luna (OpenAI API) leads on 2 rows: CLI vs API: time for a one-line answer (Total time), 0.97 s vs 4.19 s; CLI vs API: time for a one-line answer (First useful output), 0.82 s vs 3.79 s. On those rows the run ranges do not overlap; only a 95% interval is a confidence interval. The other rows are 3 unclear; each row says why. Every row ran the two sides through different routes (for example Codex CLI vs OpenAI API), so they compare route + model pairs, not models alone; the contexts name the route. Some rows rest on small samples (n = 3 at the smallest).",
      "rows": [
        {
          "metric": "CLI vs API: time for a one-line answer (Total time)",
          "aValue": 4.19,
          "bValue": 0.97,
          "unit": "seconds",
          "aDisplay": "4.19 s",
          "bDisplay": "0.97 s",
          "winner": "b",
          "basis": "The run ranges (fastest to slowest) do not overlap (GPT-6.1 Sol (Codex CLI) 3.81 s to 4.69 s; GPT-6 Luna (OpenAI API) 0.65 s to 1.50 s). A range is not a confidence interval. Samples are small (5 runs per side).",
          "studySlug": "cli-model-latency-tokens",
          "n": 5,
          "chartId": "cli-vs-api-exact-reply-latency",
          "aN": 5,
          "bN": 5,
          "aContext": "Codex CLI · effort high · fixed exact reply, 5 runs",
          "bContext": "OpenAI API · effort none · fixed exact reply, 5 runs",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            3.81,
            4.69
          ],
          "bRange": [
            0.65,
            1.5
          ]
        },
        {
          "metric": "CLI vs API: time for a one-line answer (First useful output)",
          "aValue": 3.79,
          "bValue": 0.82,
          "unit": "seconds",
          "aDisplay": "3.79 s",
          "bDisplay": "0.82 s",
          "winner": "b",
          "basis": "The run ranges (fastest to slowest) do not overlap (GPT-6.1 Sol (Codex CLI) 3.37 s to 4.30 s; GPT-6 Luna (OpenAI API) 0.51 s to 1.37 s). A range is not a confidence interval. Samples are small (5 runs per side).",
          "studySlug": "cli-model-latency-tokens",
          "n": 5,
          "chartId": "cli-vs-api-exact-reply-latency",
          "aN": 5,
          "bN": 5,
          "aContext": "Codex CLI · effort high · fixed exact reply, 5 runs",
          "bContext": "OpenAI API · effort none · fixed exact reply, 5 runs",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            3.37,
            4.3
          ],
          "bRange": [
            0.51,
            1.37
          ]
        },
        {
          "metric": "CLI vs API: time for a small coding task (Total time)",
          "aValue": 17.85,
          "bValue": 4.01,
          "unit": "seconds",
          "aDisplay": "17.9 s",
          "bDisplay": "4.01 s",
          "winner": "unclear",
          "basis": "Only 3 runs per side; the run ranges (fastest to slowest) do not overlap (GPT-6.1 Sol (Codex CLI) 17.7 s to 22.4 s; GPT-6 Luna (OpenAI API) 3.83 s to 4.35 s), but 3 runs cannot show a reliable difference. A range is not a confidence interval.",
          "studySlug": "cli-model-latency-tokens",
          "n": 3,
          "chartId": "cli-vs-api-small-coding-latency",
          "aN": 3,
          "bN": 3,
          "aContext": "Codex CLI · effort high · small coding task, 3 runs",
          "bContext": "OpenAI API · effort none · small coding task, 3 runs",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            17.68,
            22.42
          ],
          "bRange": [
            3.83,
            4.35
          ]
        },
        {
          "metric": "CLI vs API: time for a small coding task (First useful output)",
          "aValue": 17.27,
          "bValue": 0.67,
          "unit": "seconds",
          "aDisplay": "17.3 s",
          "bDisplay": "0.67 s",
          "winner": "unclear",
          "basis": "Only 3 runs per side; the run ranges (fastest to slowest) do not overlap (GPT-6.1 Sol (Codex CLI) 17.1 s to 21.9 s; GPT-6 Luna (OpenAI API) 0.62 s to 0.81 s), but 3 runs cannot show a reliable difference. A range is not a confidence interval.",
          "studySlug": "cli-model-latency-tokens",
          "n": 3,
          "chartId": "cli-vs-api-small-coding-latency",
          "aN": 3,
          "bN": 3,
          "aContext": "Codex CLI · effort high · small coding task, 3 runs",
          "bContext": "OpenAI API · effort none · small coding task, 3 runs",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            17.13,
            21.86
          ],
          "bRange": [
            0.62,
            0.81
          ]
        },
        {
          "metric": "Hidden prompt: input tokens for the same one-line request",
          "aValue": 19555,
          "bValue": 17,
          "unit": "tokens",
          "aDisplay": "19,555",
          "bDisplay": "17",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "cli-model-latency-tokens",
          "n": 5,
          "chartId": "cli-vs-api-prompt-overhead",
          "aN": 5,
          "bN": 5,
          "aContext": "Codex CLI · effort high · short fixed tasks",
          "bContext": "OpenAI API · effort none · short fixed tasks"
        }
      ]
    },
    {
      "slug": "gpt-6-1-sol-openai-api-vs-gpt-6-luna-codex-cli",
      "a": "gpt-6-1-sol-openai-api",
      "b": "gpt-6-luna-codex-cli",
      "title": "GPT-6.1 Sol (OpenAI API) vs GPT-6 Luna (Codex CLI)",
      "seoTitle": "GPT-6.1 Sol (OpenAI API) vs GPT-6 Luna (Codex CLI)",
      "description": "GPT-6.1 Sol (OpenAI API) vs GPT-6 Luna (Codex CLI): 5 measured metrics from one study, with sample sizes, intervals and every failure counted.",
      "verdict": "GPT-6.1 Sol (OpenAI API) and GPT-6 Luna (Codex CLI) share 5 measured metrics from one study. GPT-6.1 Sol (OpenAI API) leads on 2 rows: CLI vs API: time for a one-line answer (Total time), 1.52 s vs 3.19 s; CLI vs API: time for a one-line answer (First useful output), 1.34 s vs 2.79 s. On those rows the run ranges do not overlap; only a 95% interval is a confidence interval. The other rows are 3 unclear; each row says why. Every row ran the two sides through different routes (for example OpenAI API vs Codex CLI), so they compare route + model pairs, not models alone; the contexts name the route. Some rows rest on small samples (n = 3 at the smallest).",
      "rows": [
        {
          "metric": "CLI vs API: time for a one-line answer (Total time)",
          "aValue": 1.52,
          "bValue": 3.19,
          "unit": "seconds",
          "aDisplay": "1.52 s",
          "bDisplay": "3.19 s",
          "winner": "a",
          "basis": "The run ranges (fastest to slowest) do not overlap (GPT-6.1 Sol (OpenAI API) 1.35 s to 2.23 s; GPT-6 Luna (Codex CLI) 2.88 s to 3.83 s). A range is not a confidence interval. Samples are small (5 runs per side).",
          "studySlug": "cli-model-latency-tokens",
          "n": 5,
          "chartId": "cli-vs-api-exact-reply-latency",
          "aN": 5,
          "bN": 5,
          "aContext": "OpenAI API · effort high · fixed exact reply, 5 runs",
          "bContext": "Codex CLI · effort none · fixed exact reply, 5 runs",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            1.35,
            2.23
          ],
          "bRange": [
            2.88,
            3.83
          ]
        },
        {
          "metric": "CLI vs API: time for a one-line answer (First useful output)",
          "aValue": 1.34,
          "bValue": 2.79,
          "unit": "seconds",
          "aDisplay": "1.34 s",
          "bDisplay": "2.79 s",
          "winner": "a",
          "basis": "The run ranges (fastest to slowest) do not overlap (GPT-6.1 Sol (OpenAI API) 1.26 s to 2.12 s; GPT-6 Luna (Codex CLI) 2.46 s to 3.42 s). A range is not a confidence interval. Samples are small (5 runs per side).",
          "studySlug": "cli-model-latency-tokens",
          "n": 5,
          "chartId": "cli-vs-api-exact-reply-latency",
          "aN": 5,
          "bN": 5,
          "aContext": "OpenAI API · effort high · fixed exact reply, 5 runs",
          "bContext": "Codex CLI · effort none · fixed exact reply, 5 runs",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            1.26,
            2.12
          ],
          "bRange": [
            2.46,
            3.42
          ]
        },
        {
          "metric": "CLI vs API: time for a small coding task (Total time)",
          "aValue": 9.56,
          "bValue": 9.23,
          "unit": "seconds",
          "aDisplay": "9.56 s",
          "bDisplay": "9.23 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (GPT-6.1 Sol (OpenAI API) 9.44 s to 10.9 s; GPT-6 Luna (Codex CLI) 8.99 s to 11.7 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "cli-model-latency-tokens",
          "n": 3,
          "chartId": "cli-vs-api-small-coding-latency",
          "aN": 3,
          "bN": 3,
          "aContext": "OpenAI API · effort high · small coding task, 3 runs",
          "bContext": "Codex CLI · effort none · small coding task, 3 runs",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            9.44,
            10.94
          ],
          "bRange": [
            8.99,
            11.68
          ]
        },
        {
          "metric": "CLI vs API: time for a small coding task (First useful output)",
          "aValue": 5.31,
          "bValue": 8.68,
          "unit": "seconds",
          "aDisplay": "5.31 s",
          "bDisplay": "8.68 s",
          "winner": "unclear",
          "basis": "Only 3 runs per side; the run ranges (fastest to slowest) do not overlap (GPT-6.1 Sol (OpenAI API) 4.99 s to 6.42 s; GPT-6 Luna (Codex CLI) 8.27 s to 11.0 s), but 3 runs cannot show a reliable difference. A range is not a confidence interval.",
          "studySlug": "cli-model-latency-tokens",
          "n": 3,
          "chartId": "cli-vs-api-small-coding-latency",
          "aN": 3,
          "bN": 3,
          "aContext": "OpenAI API · effort high · small coding task, 3 runs",
          "bContext": "Codex CLI · effort none · small coding task, 3 runs",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            4.99,
            6.42
          ],
          "bRange": [
            8.27,
            11.01
          ]
        },
        {
          "metric": "Hidden prompt: input tokens for the same one-line request",
          "aValue": 17,
          "bValue": 18859,
          "unit": "tokens",
          "aDisplay": "17",
          "bDisplay": "18,859",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "cli-model-latency-tokens",
          "n": 5,
          "chartId": "cli-vs-api-prompt-overhead",
          "aN": 5,
          "bN": 5,
          "aContext": "OpenAI API · effort high · short fixed tasks",
          "bContext": "Codex CLI · effort none · short fixed tasks"
        }
      ]
    },
    {
      "slug": "gpt-6-1-sol-openai-api-vs-gpt-6-luna-openai-api",
      "a": "gpt-6-1-sol-openai-api",
      "b": "gpt-6-luna-openai-api",
      "title": "GPT-6.1 Sol (OpenAI API) vs GPT-6 Luna (OpenAI API)",
      "seoTitle": "GPT-6.1 Sol (OpenAI API) vs GPT-6 Luna (OpenAI API)",
      "description": "GPT-6.1 Sol (OpenAI API) vs GPT-6 Luna (OpenAI API): 5 measured metrics from one study, with sample sizes, intervals and every failure counted.",
      "verdict": "GPT-6.1 Sol (OpenAI API) and GPT-6 Luna (OpenAI API) share 5 measured metrics from one study. No row separates them: every interval or run range overlaps, too few runs were recorded, no interval was recorded, or more is not better for that metric. The rows are 1 tie and 4 unclear; each row says why. Some rows rest on small samples (n = 3 at the smallest).",
      "rows": [
        {
          "metric": "CLI vs API: time for a one-line answer (Total time)",
          "aValue": 1.52,
          "bValue": 0.97,
          "unit": "seconds",
          "aDisplay": "1.52 s",
          "bDisplay": "0.97 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (GPT-6.1 Sol (OpenAI API) 1.35 s to 2.23 s; GPT-6 Luna (OpenAI API) 0.65 s to 1.50 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "cli-model-latency-tokens",
          "n": 5,
          "chartId": "cli-vs-api-exact-reply-latency",
          "aN": 5,
          "bN": 5,
          "aContext": "OpenAI API · effort high · fixed exact reply, 5 runs",
          "bContext": "OpenAI API · effort none · fixed exact reply, 5 runs",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            1.35,
            2.23
          ],
          "bRange": [
            0.65,
            1.5
          ]
        },
        {
          "metric": "CLI vs API: time for a one-line answer (First useful output)",
          "aValue": 1.34,
          "bValue": 0.82,
          "unit": "seconds",
          "aDisplay": "1.34 s",
          "bDisplay": "0.82 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (GPT-6.1 Sol (OpenAI API) 1.26 s to 2.12 s; GPT-6 Luna (OpenAI API) 0.51 s to 1.37 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "cli-model-latency-tokens",
          "n": 5,
          "chartId": "cli-vs-api-exact-reply-latency",
          "aN": 5,
          "bN": 5,
          "aContext": "OpenAI API · effort high · fixed exact reply, 5 runs",
          "bContext": "OpenAI API · effort none · fixed exact reply, 5 runs",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            1.26,
            2.12
          ],
          "bRange": [
            0.51,
            1.37
          ]
        },
        {
          "metric": "CLI vs API: time for a small coding task (Total time)",
          "aValue": 9.56,
          "bValue": 4.01,
          "unit": "seconds",
          "aDisplay": "9.56 s",
          "bDisplay": "4.01 s",
          "winner": "unclear",
          "basis": "Only 3 runs per side; the run ranges (fastest to slowest) do not overlap (GPT-6.1 Sol (OpenAI API) 9.44 s to 10.9 s; GPT-6 Luna (OpenAI API) 3.83 s to 4.35 s), but 3 runs cannot show a reliable difference. A range is not a confidence interval.",
          "studySlug": "cli-model-latency-tokens",
          "n": 3,
          "chartId": "cli-vs-api-small-coding-latency",
          "aN": 3,
          "bN": 3,
          "aContext": "OpenAI API · effort high · small coding task, 3 runs",
          "bContext": "OpenAI API · effort none · small coding task, 3 runs",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            9.44,
            10.94
          ],
          "bRange": [
            3.83,
            4.35
          ]
        },
        {
          "metric": "CLI vs API: time for a small coding task (First useful output)",
          "aValue": 5.31,
          "bValue": 0.67,
          "unit": "seconds",
          "aDisplay": "5.31 s",
          "bDisplay": "0.67 s",
          "winner": "unclear",
          "basis": "Only 3 runs per side; the run ranges (fastest to slowest) do not overlap (GPT-6.1 Sol (OpenAI API) 4.99 s to 6.42 s; GPT-6 Luna (OpenAI API) 0.62 s to 0.81 s), but 3 runs cannot show a reliable difference. A range is not a confidence interval.",
          "studySlug": "cli-model-latency-tokens",
          "n": 3,
          "chartId": "cli-vs-api-small-coding-latency",
          "aN": 3,
          "bN": 3,
          "aContext": "OpenAI API · effort high · small coding task, 3 runs",
          "bContext": "OpenAI API · effort none · small coding task, 3 runs",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            4.99,
            6.42
          ],
          "bRange": [
            0.62,
            0.81
          ]
        },
        {
          "metric": "Hidden prompt: input tokens for the same one-line request",
          "aValue": 17,
          "bValue": 17,
          "unit": "tokens",
          "aDisplay": "17",
          "bDisplay": "17",
          "winner": "tie",
          "basis": "Same value. More or fewer is not better by itself for this metric.",
          "studySlug": "cli-model-latency-tokens",
          "n": 5,
          "chartId": "cli-vs-api-prompt-overhead",
          "aN": 5,
          "bN": 5,
          "aContext": "OpenAI API · effort high · short fixed tasks",
          "bContext": "OpenAI API · effort none · short fixed tasks"
        }
      ]
    },
    {
      "slug": "gpt-6-luna-codex-cli-vs-gpt-6-luna-openai-api",
      "a": "gpt-6-luna-codex-cli",
      "b": "gpt-6-luna-openai-api",
      "title": "GPT-6 Luna (Codex CLI) vs GPT-6 Luna (OpenAI API)",
      "seoTitle": "GPT-6 Luna (Codex CLI) vs GPT-6 Luna (OpenAI API)",
      "description": "GPT-6 Luna (Codex CLI) vs GPT-6 Luna (OpenAI API): 5 measured metrics from one study, with sample sizes, intervals and every failure counted.",
      "verdict": "GPT-6 Luna (Codex CLI) and GPT-6 Luna (OpenAI API) share 5 measured metrics from one study. GPT-6 Luna (OpenAI API) leads on 2 rows: CLI vs API: time for a one-line answer (Total time), 0.97 s vs 3.19 s; CLI vs API: time for a one-line answer (First useful output), 0.82 s vs 2.79 s. On those rows the run ranges do not overlap; only a 95% interval is a confidence interval. The other rows are 3 unclear; each row says why. Every row ran the two sides through different routes (for example Codex CLI vs OpenAI API), so they compare route + model pairs, not models alone; the contexts name the route. Some rows rest on small samples (n = 3 at the smallest).",
      "rows": [
        {
          "metric": "CLI vs API: time for a one-line answer (Total time)",
          "aValue": 3.19,
          "bValue": 0.97,
          "unit": "seconds",
          "aDisplay": "3.19 s",
          "bDisplay": "0.97 s",
          "winner": "b",
          "basis": "The run ranges (fastest to slowest) do not overlap (GPT-6 Luna (Codex CLI) 2.88 s to 3.83 s; GPT-6 Luna (OpenAI API) 0.65 s to 1.50 s). A range is not a confidence interval. Samples are small (5 runs per side).",
          "studySlug": "cli-model-latency-tokens",
          "n": 5,
          "chartId": "cli-vs-api-exact-reply-latency",
          "aN": 5,
          "bN": 5,
          "aContext": "Codex CLI · effort none · fixed exact reply, 5 runs",
          "bContext": "OpenAI API · effort none · fixed exact reply, 5 runs",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            2.88,
            3.83
          ],
          "bRange": [
            0.65,
            1.5
          ]
        },
        {
          "metric": "CLI vs API: time for a one-line answer (First useful output)",
          "aValue": 2.79,
          "bValue": 0.82,
          "unit": "seconds",
          "aDisplay": "2.79 s",
          "bDisplay": "0.82 s",
          "winner": "b",
          "basis": "The run ranges (fastest to slowest) do not overlap (GPT-6 Luna (Codex CLI) 2.46 s to 3.42 s; GPT-6 Luna (OpenAI API) 0.51 s to 1.37 s). A range is not a confidence interval. Samples are small (5 runs per side).",
          "studySlug": "cli-model-latency-tokens",
          "n": 5,
          "chartId": "cli-vs-api-exact-reply-latency",
          "aN": 5,
          "bN": 5,
          "aContext": "Codex CLI · effort none · fixed exact reply, 5 runs",
          "bContext": "OpenAI API · effort none · fixed exact reply, 5 runs",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            2.46,
            3.42
          ],
          "bRange": [
            0.51,
            1.37
          ]
        },
        {
          "metric": "CLI vs API: time for a small coding task (Total time)",
          "aValue": 9.23,
          "bValue": 4.01,
          "unit": "seconds",
          "aDisplay": "9.23 s",
          "bDisplay": "4.01 s",
          "winner": "unclear",
          "basis": "Only 3 runs per side; the run ranges (fastest to slowest) do not overlap (GPT-6 Luna (Codex CLI) 8.99 s to 11.7 s; GPT-6 Luna (OpenAI API) 3.83 s to 4.35 s), but 3 runs cannot show a reliable difference. A range is not a confidence interval.",
          "studySlug": "cli-model-latency-tokens",
          "n": 3,
          "chartId": "cli-vs-api-small-coding-latency",
          "aN": 3,
          "bN": 3,
          "aContext": "Codex CLI · effort none · small coding task, 3 runs",
          "bContext": "OpenAI API · effort none · small coding task, 3 runs",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            8.99,
            11.68
          ],
          "bRange": [
            3.83,
            4.35
          ]
        },
        {
          "metric": "CLI vs API: time for a small coding task (First useful output)",
          "aValue": 8.68,
          "bValue": 0.67,
          "unit": "seconds",
          "aDisplay": "8.68 s",
          "bDisplay": "0.67 s",
          "winner": "unclear",
          "basis": "Only 3 runs per side; the run ranges (fastest to slowest) do not overlap (GPT-6 Luna (Codex CLI) 8.27 s to 11.0 s; GPT-6 Luna (OpenAI API) 0.62 s to 0.81 s), but 3 runs cannot show a reliable difference. A range is not a confidence interval.",
          "studySlug": "cli-model-latency-tokens",
          "n": 3,
          "chartId": "cli-vs-api-small-coding-latency",
          "aN": 3,
          "bN": 3,
          "aContext": "Codex CLI · effort none · small coding task, 3 runs",
          "bContext": "OpenAI API · effort none · small coding task, 3 runs",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            8.27,
            11.01
          ],
          "bRange": [
            0.62,
            0.81
          ]
        },
        {
          "metric": "Hidden prompt: input tokens for the same one-line request",
          "aValue": 18859,
          "bValue": 17,
          "unit": "tokens",
          "aDisplay": "18,859",
          "bDisplay": "17",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "cli-model-latency-tokens",
          "n": 5,
          "chartId": "cli-vs-api-prompt-overhead",
          "aN": 5,
          "bN": 5,
          "aContext": "Codex CLI · effort none · short fixed tasks",
          "bContext": "OpenAI API · effort none · short fixed tasks"
        }
      ]
    },
    {
      "slug": "novita-vs-baseten",
      "a": "novita",
      "b": "baseten",
      "title": "Novita AI vs Baseten",
      "seoTitle": "Novita AI vs Baseten: price per model",
      "description": "Novita AI vs Baseten: 5 list prices per million tokens for the same models, as reported by OpenRouter’s public API, with the gap stated.",
      "verdict": "Novita AI and Baseten share 5 reported list prices from one study. No row names a winner: a reported list price has no interval, so each gap is stated, not ranked. Prices change often; the snapshot date is in each row’s context. The rows are 5 unclear; each row says why.",
      "rows": [
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Input)",
          "aValue": 0.05,
          "bValue": 0.1,
          "unit": "usd",
          "aDisplay": "$0.050",
          "bDisplay": "$0.10",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.050 vs $0.10, 2.0x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Output)",
          "aValue": 0.25,
          "bValue": 0.5,
          "unit": "usd",
          "aDisplay": "$0.25",
          "bDisplay": "$0.50",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.25 vs $0.50, 2.0x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "GLM 5.3: price per million tokens by provider (Input)",
          "aValue": 0.42,
          "bValue": 1.4,
          "unit": "usd",
          "aDisplay": "$0.42",
          "bDisplay": "$1.40",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.42 vs $1.40, 3.3x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "aContext": "fp8 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp4 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "GLM 5.3: price per million tokens by provider (Output)",
          "aValue": 1.32,
          "bValue": 4.4,
          "unit": "usd",
          "aDisplay": "$1.32",
          "bDisplay": "$4.40",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($1.32 vs $4.40, 3.3x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "aContext": "fp8 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp4 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "GLM 5.3: price per million tokens by provider (Cache read)",
          "aValue": 0.078,
          "bValue": 0.14,
          "unit": "usd",
          "aDisplay": "$0.078",
          "bDisplay": "$0.14",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.078 vs $0.14, 1.8x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "aContext": "fp8 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp4 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "together-vs-cloudflare",
      "a": "together",
      "b": "cloudflare",
      "title": "Together AI vs Cloudflare Workers AI",
      "seoTitle": "Together AI vs Cloudflare Workers AI: price per model",
      "description": "Together AI vs Cloudflare Workers AI: 5 list prices per million tokens for the same models, as reported by OpenRouter’s public API, with the gap stated.",
      "verdict": "Together AI and Cloudflare Workers AI share 5 reported list prices from one study. No row names a winner: a reported list price has no interval, so each gap is stated, not ranked. Prices change often; the snapshot date is in each row’s context. The rows are 3 ties and 2 unclear; each row says why.",
      "rows": [
        {
          "metric": "Llama 3.3 70B Instruct: price per million tokens by provider (Input)",
          "aValue": 1.04,
          "bValue": 0.293,
          "unit": "usd",
          "aDisplay": "$1.04",
          "bDisplay": "$0.29",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($1.04 vs $0.29, 3.5x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-llama-3-3-70b-instruct",
          "aContext": "Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Llama 3.3 70B Instruct: price per million tokens by provider (Output)",
          "aValue": 1.04,
          "bValue": 2.253,
          "unit": "usd",
          "aDisplay": "$1.04",
          "bDisplay": "$2.25",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($1.04 vs $2.25, 2.2x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-llama-3-3-70b-instruct",
          "aContext": "Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "GLM 5.3: price per million tokens by provider (Input)",
          "aValue": 1.4,
          "bValue": 1.4,
          "unit": "usd",
          "aDisplay": "$1.40",
          "bDisplay": "$1.40",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "aContext": "GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "GLM 5.3: price per million tokens by provider (Output)",
          "aValue": 4.4,
          "bValue": 4.4,
          "unit": "usd",
          "aDisplay": "$4.40",
          "bDisplay": "$4.40",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "aContext": "GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "GLM 5.3: price per million tokens by provider (Cache read)",
          "aValue": 0.26,
          "bValue": 0.26,
          "unit": "usd",
          "aDisplay": "$0.26",
          "bDisplay": "$0.26",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "aContext": "GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "together-vs-siliconflow",
      "a": "together",
      "b": "siliconflow",
      "title": "Together AI vs SiliconFlow",
      "seoTitle": "Together AI vs SiliconFlow: price per model",
      "description": "Together AI vs SiliconFlow: 5 list prices per million tokens for the same models, as reported by OpenRouter’s public API, with the gap stated.",
      "verdict": "Together AI and SiliconFlow share 5 reported list prices from one study. No row names a winner: a reported list price has no interval, so each gap is stated, not ranked. Prices change often; the snapshot date is in each row’s context. The rows are 2 ties and 3 unclear; each row says why.",
      "rows": [
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Input)",
          "aValue": 0.15,
          "bValue": 0.15,
          "unit": "usd",
          "aDisplay": "$0.15",
          "bDisplay": "$0.15",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Output)",
          "aValue": 0.6,
          "bValue": 0.6,
          "unit": "usd",
          "aDisplay": "$0.60",
          "bDisplay": "$0.60",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "GLM 5.3: price per million tokens by provider (Input)",
          "aValue": 1.4,
          "bValue": 0.7,
          "unit": "usd",
          "aDisplay": "$1.40",
          "bDisplay": "$0.70",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($1.40 vs $0.70, 2.0x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "aContext": "GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "GLM 5.3: price per million tokens by provider (Output)",
          "aValue": 4.4,
          "bValue": 2.2,
          "unit": "usd",
          "aDisplay": "$4.40",
          "bDisplay": "$2.20",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($4.40 vs $2.20, 2.0x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "aContext": "GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "GLM 5.3: price per million tokens by provider (Cache read)",
          "aValue": 0.26,
          "bValue": 0.13,
          "unit": "usd",
          "aDisplay": "$0.26",
          "bDisplay": "$0.13",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.26 vs $0.13, 2.0x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "aContext": "GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "deepinfra-vs-nebius",
      "a": "deepinfra",
      "b": "nebius",
      "title": "DeepInfra vs Nebius",
      "seoTitle": "DeepInfra vs Nebius: price per model",
      "description": "DeepInfra vs Nebius: 4 list prices per million tokens for the same models, as reported by OpenRouter’s public API, with the gap stated.",
      "verdict": "DeepInfra and Nebius share 4 reported list prices from one study. No row names a winner: a reported list price has no interval, so each gap is stated, not ranked. Prices change often; the snapshot date is in each row’s context. The rows are 4 unclear; each row says why.",
      "rows": [
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Input)",
          "aValue": 0.037,
          "bValue": 0.15,
          "unit": "usd",
          "aDisplay": "$0.037",
          "bDisplay": "$0.15",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.037 vs $0.15, 4.1x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "bf16 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Output)",
          "aValue": 0.17,
          "bValue": 0.6,
          "unit": "usd",
          "aDisplay": "$0.17",
          "bDisplay": "$0.60",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.17 vs $0.60, 3.5x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "bf16 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "GLM 5.3: price per million tokens by provider (Input)",
          "aValue": 0.5625,
          "bValue": 1.4,
          "unit": "usd",
          "aDisplay": "$0.56",
          "bDisplay": "$1.40",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.56 vs $1.40, 2.5x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "aContext": "fp4 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp4 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "GLM 5.3: price per million tokens by provider (Output)",
          "aValue": 2.5,
          "bValue": 4.4,
          "unit": "usd",
          "aDisplay": "$2.50",
          "bDisplay": "$4.40",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($2.50 vs $4.40, 1.8x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "aContext": "fp4 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp4 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "deepinfra-vs-sambanova",
      "a": "deepinfra",
      "b": "sambanova",
      "title": "DeepInfra vs SambaNova",
      "seoTitle": "DeepInfra vs SambaNova: price per model",
      "description": "DeepInfra vs SambaNova: 4 list prices per million tokens for the same models, as reported by OpenRouter’s public API, with the gap stated.",
      "verdict": "DeepInfra and SambaNova share 4 reported list prices from one study. No row names a winner: a reported list price has no interval, so each gap is stated, not ranked. Prices change often; the snapshot date is in each row’s context. The rows are 4 unclear; each row says why.",
      "rows": [
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Input)",
          "aValue": 0.037,
          "bValue": 0.14,
          "unit": "usd",
          "aDisplay": "$0.037",
          "bDisplay": "$0.14",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.037 vs $0.14, 3.8x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "bf16 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Output)",
          "aValue": 0.17,
          "bValue": 0.95,
          "unit": "usd",
          "aDisplay": "$0.17",
          "bDisplay": "$0.95",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.17 vs $0.95, 5.6x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "bf16 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Llama 3.3 70B Instruct: price per million tokens by provider (Input)",
          "aValue": 0.1,
          "bValue": 0.45,
          "unit": "usd",
          "aDisplay": "$0.10",
          "bDisplay": "$0.45",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.10 vs $0.45, 4.5x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-llama-3-3-70b-instruct",
          "aContext": "fp8 · Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Llama 3.3 70B Instruct: price per million tokens by provider (Output)",
          "aValue": 0.32,
          "bValue": 0.9,
          "unit": "usd",
          "aDisplay": "$0.32",
          "bDisplay": "$0.90",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.32 vs $0.90, 2.8x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-llama-3-3-70b-instruct",
          "aContext": "fp8 · Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "google-vertex-vs-deepinfra",
      "a": "google-vertex",
      "b": "deepinfra",
      "title": "Google Vertex AI vs DeepInfra",
      "seoTitle": "Google Vertex AI vs DeepInfra: price per model",
      "description": "Google Vertex AI vs DeepInfra: 4 list prices per million tokens for the same models, as reported by OpenRouter’s public API, with the gap stated.",
      "verdict": "Google Vertex AI and DeepInfra share 4 reported list prices from one study. No row names a winner: a reported list price has no interval, so each gap is stated, not ranked. Prices change often; the snapshot date is in each row’s context. The rows are 4 unclear; each row says why.",
      "rows": [
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Input)",
          "aValue": 0.09,
          "bValue": 0.037,
          "unit": "usd",
          "aDisplay": "$0.090",
          "bDisplay": "$0.037",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.090 vs $0.037, 2.4x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "bf16 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Output)",
          "aValue": 0.36,
          "bValue": 0.17,
          "unit": "usd",
          "aDisplay": "$0.36",
          "bDisplay": "$0.17",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.36 vs $0.17, 2.1x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "bf16 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Llama 3.3 70B Instruct: price per million tokens by provider (Input)",
          "aValue": 0.72,
          "bValue": 0.1,
          "unit": "usd",
          "aDisplay": "$0.72",
          "bDisplay": "$0.10",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.72 vs $0.10, 7.2x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-llama-3-3-70b-instruct",
          "aContext": "Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Llama 3.3 70B Instruct: price per million tokens by provider (Output)",
          "aValue": 0.72,
          "bValue": 0.32,
          "unit": "usd",
          "aDisplay": "$0.72",
          "bDisplay": "$0.32",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.72 vs $0.32, 2.3x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-llama-3-3-70b-instruct",
          "aContext": "Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "google-vertex-vs-groq",
      "a": "google-vertex",
      "b": "groq",
      "title": "Google Vertex AI vs Groq",
      "seoTitle": "Google Vertex AI vs Groq: price per model",
      "description": "Google Vertex AI vs Groq: 4 list prices per million tokens for the same models, as reported by OpenRouter’s public API, with the gap stated.",
      "verdict": "Google Vertex AI and Groq share 4 reported list prices from one study. No row names a winner: a reported list price has no interval, so each gap is stated, not ranked. Prices change often; the snapshot date is in each row’s context. The rows are 4 unclear; each row says why.",
      "rows": [
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Input)",
          "aValue": 0.09,
          "bValue": 0.15,
          "unit": "usd",
          "aDisplay": "$0.090",
          "bDisplay": "$0.15",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.090 vs $0.15, 1.7x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Output)",
          "aValue": 0.36,
          "bValue": 0.6,
          "unit": "usd",
          "aDisplay": "$0.36",
          "bDisplay": "$0.60",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.36 vs $0.60, 1.7x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Llama 3.3 70B Instruct: price per million tokens by provider (Input)",
          "aValue": 0.72,
          "bValue": 0.59,
          "unit": "usd",
          "aDisplay": "$0.72",
          "bDisplay": "$0.59",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.72 vs $0.59, 1.2x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-llama-3-3-70b-instruct",
          "aContext": "Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Llama 3.3 70B Instruct: price per million tokens by provider (Output)",
          "aValue": 0.72,
          "bValue": 0.79,
          "unit": "usd",
          "aDisplay": "$0.72",
          "bDisplay": "$0.79",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.72 vs $0.79, 1.1x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-llama-3-3-70b-instruct",
          "aContext": "Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "google-vertex-vs-novita",
      "a": "google-vertex",
      "b": "novita",
      "title": "Google Vertex AI vs Novita AI",
      "seoTitle": "Google Vertex AI vs Novita AI: price per model",
      "description": "Google Vertex AI vs Novita AI: 4 list prices per million tokens for the same models, as reported by OpenRouter’s public API, with the gap stated.",
      "verdict": "Google Vertex AI and Novita AI share 4 reported list prices from one study. No row names a winner: a reported list price has no interval, so each gap is stated, not ranked. Prices change often; the snapshot date is in each row’s context. The rows are 4 unclear; each row says why.",
      "rows": [
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Input)",
          "aValue": 0.09,
          "bValue": 0.05,
          "unit": "usd",
          "aDisplay": "$0.090",
          "bDisplay": "$0.050",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.090 vs $0.050, 1.8x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Output)",
          "aValue": 0.36,
          "bValue": 0.25,
          "unit": "usd",
          "aDisplay": "$0.36",
          "bDisplay": "$0.25",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.36 vs $0.25, 1.4x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Llama 3.3 70B Instruct: price per million tokens by provider (Input)",
          "aValue": 0.72,
          "bValue": 0.135,
          "unit": "usd",
          "aDisplay": "$0.72",
          "bDisplay": "$0.14",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.72 vs $0.14, 5.3x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-llama-3-3-70b-instruct",
          "aContext": "Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "bf16 · Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Llama 3.3 70B Instruct: price per million tokens by provider (Output)",
          "aValue": 0.72,
          "bValue": 0.4,
          "unit": "usd",
          "aDisplay": "$0.72",
          "bDisplay": "$0.40",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.72 vs $0.40, 1.8x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-llama-3-3-70b-instruct",
          "aContext": "Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "bf16 · Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "google-vertex-vs-parasail",
      "a": "google-vertex",
      "b": "parasail",
      "title": "Google Vertex AI vs Parasail",
      "seoTitle": "Google Vertex AI vs Parasail: price per model",
      "description": "Google Vertex AI vs Parasail: 4 list prices per million tokens for the same models, as reported by OpenRouter’s public API, with the gap stated.",
      "verdict": "Google Vertex AI and Parasail share 4 reported list prices from one study. No row names a winner: a reported list price has no interval, so each gap is stated, not ranked. Prices change often; the snapshot date is in each row’s context. The rows are 4 unclear; each row says why.",
      "rows": [
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Input)",
          "aValue": 0.09,
          "bValue": 0.1,
          "unit": "usd",
          "aDisplay": "$0.090",
          "bDisplay": "$0.10",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.090 vs $0.10, 1.1x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Output)",
          "aValue": 0.36,
          "bValue": 0.75,
          "unit": "usd",
          "aDisplay": "$0.36",
          "bDisplay": "$0.75",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.36 vs $0.75, 2.1x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Llama 3.3 70B Instruct: price per million tokens by provider (Input)",
          "aValue": 0.72,
          "bValue": 0.22,
          "unit": "usd",
          "aDisplay": "$0.72",
          "bDisplay": "$0.22",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.72 vs $0.22, 3.3x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-llama-3-3-70b-instruct",
          "aContext": "Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Llama 3.3 70B Instruct: price per million tokens by provider (Output)",
          "aValue": 0.72,
          "bValue": 0.5,
          "unit": "usd",
          "aDisplay": "$0.72",
          "bDisplay": "$0.50",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.72 vs $0.50, 1.4x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-llama-3-3-70b-instruct",
          "aContext": "Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "google-vertex-vs-sambanova",
      "a": "google-vertex",
      "b": "sambanova",
      "title": "Google Vertex AI vs SambaNova",
      "seoTitle": "Google Vertex AI vs SambaNova: price per model",
      "description": "Google Vertex AI vs SambaNova: 4 list prices per million tokens for the same models, as reported by OpenRouter’s public API, with the gap stated.",
      "verdict": "Google Vertex AI and SambaNova share 4 reported list prices from one study. No row names a winner: a reported list price has no interval, so each gap is stated, not ranked. Prices change often; the snapshot date is in each row’s context. The rows are 4 unclear; each row says why.",
      "rows": [
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Input)",
          "aValue": 0.09,
          "bValue": 0.14,
          "unit": "usd",
          "aDisplay": "$0.090",
          "bDisplay": "$0.14",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.090 vs $0.14, 1.6x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Output)",
          "aValue": 0.36,
          "bValue": 0.95,
          "unit": "usd",
          "aDisplay": "$0.36",
          "bDisplay": "$0.95",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.36 vs $0.95, 2.6x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Llama 3.3 70B Instruct: price per million tokens by provider (Input)",
          "aValue": 0.72,
          "bValue": 0.45,
          "unit": "usd",
          "aDisplay": "$0.72",
          "bDisplay": "$0.45",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.72 vs $0.45, 1.6x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-llama-3-3-70b-instruct",
          "aContext": "Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Llama 3.3 70B Instruct: price per million tokens by provider (Output)",
          "aValue": 0.72,
          "bValue": 0.9,
          "unit": "usd",
          "aDisplay": "$0.72",
          "bDisplay": "$0.90",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.72 vs $0.90, 1.3x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-llama-3-3-70b-instruct",
          "aContext": "Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "google-vertex-vs-together",
      "a": "google-vertex",
      "b": "together",
      "title": "Google Vertex AI vs Together AI",
      "seoTitle": "Google Vertex AI vs Together AI: price per model",
      "description": "Google Vertex AI vs Together AI: 4 list prices per million tokens for the same models, as reported by OpenRouter’s public API, with the gap stated.",
      "verdict": "Google Vertex AI and Together AI share 4 reported list prices from one study. No row names a winner: a reported list price has no interval, so each gap is stated, not ranked. Prices change often; the snapshot date is in each row’s context. The rows are 4 unclear; each row says why.",
      "rows": [
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Input)",
          "aValue": 0.09,
          "bValue": 0.15,
          "unit": "usd",
          "aDisplay": "$0.090",
          "bDisplay": "$0.15",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.090 vs $0.15, 1.7x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Output)",
          "aValue": 0.36,
          "bValue": 0.6,
          "unit": "usd",
          "aDisplay": "$0.36",
          "bDisplay": "$0.60",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.36 vs $0.60, 1.7x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Llama 3.3 70B Instruct: price per million tokens by provider (Input)",
          "aValue": 0.72,
          "bValue": 1.04,
          "unit": "usd",
          "aDisplay": "$0.72",
          "bDisplay": "$1.04",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.72 vs $1.04, 1.4x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-llama-3-3-70b-instruct",
          "aContext": "Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Llama 3.3 70B Instruct: price per million tokens by provider (Output)",
          "aValue": 0.72,
          "bValue": 1.04,
          "unit": "usd",
          "aDisplay": "$0.72",
          "bDisplay": "$1.04",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.72 vs $1.04, 1.4x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-llama-3-3-70b-instruct",
          "aContext": "Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "groq-vs-deepinfra",
      "a": "groq",
      "b": "deepinfra",
      "title": "Groq vs DeepInfra",
      "seoTitle": "Groq vs DeepInfra: price per model",
      "description": "Groq vs DeepInfra: 4 list prices per million tokens for the same models, as reported by OpenRouter’s public API, with the gap stated.",
      "verdict": "Groq and DeepInfra share 4 reported list prices from one study. No row names a winner: a reported list price has no interval, so each gap is stated, not ranked. Prices change often; the snapshot date is in each row’s context. The rows are 4 unclear; each row says why.",
      "rows": [
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Input)",
          "aValue": 0.15,
          "bValue": 0.037,
          "unit": "usd",
          "aDisplay": "$0.15",
          "bDisplay": "$0.037",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.15 vs $0.037, 4.1x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "bf16 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Output)",
          "aValue": 0.6,
          "bValue": 0.17,
          "unit": "usd",
          "aDisplay": "$0.60",
          "bDisplay": "$0.17",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.60 vs $0.17, 3.5x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "bf16 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Llama 3.3 70B Instruct: price per million tokens by provider (Input)",
          "aValue": 0.59,
          "bValue": 0.1,
          "unit": "usd",
          "aDisplay": "$0.59",
          "bDisplay": "$0.10",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.59 vs $0.10, 5.9x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-llama-3-3-70b-instruct",
          "aContext": "Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Llama 3.3 70B Instruct: price per million tokens by provider (Output)",
          "aValue": 0.79,
          "bValue": 0.32,
          "unit": "usd",
          "aDisplay": "$0.79",
          "bDisplay": "$0.32",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.79 vs $0.32, 2.5x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-llama-3-3-70b-instruct",
          "aContext": "Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "groq-vs-novita",
      "a": "groq",
      "b": "novita",
      "title": "Groq vs Novita AI",
      "seoTitle": "Groq vs Novita AI: price per model",
      "description": "Groq vs Novita AI: 4 list prices per million tokens for the same models, as reported by OpenRouter’s public API, with the gap stated.",
      "verdict": "Groq and Novita AI share 4 reported list prices from one study. No row names a winner: a reported list price has no interval, so each gap is stated, not ranked. Prices change often; the snapshot date is in each row’s context. The rows are 4 unclear; each row says why.",
      "rows": [
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Input)",
          "aValue": 0.15,
          "bValue": 0.05,
          "unit": "usd",
          "aDisplay": "$0.15",
          "bDisplay": "$0.050",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.15 vs $0.050, 3.0x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Output)",
          "aValue": 0.6,
          "bValue": 0.25,
          "unit": "usd",
          "aDisplay": "$0.60",
          "bDisplay": "$0.25",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.60 vs $0.25, 2.4x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Llama 3.3 70B Instruct: price per million tokens by provider (Input)",
          "aValue": 0.59,
          "bValue": 0.135,
          "unit": "usd",
          "aDisplay": "$0.59",
          "bDisplay": "$0.14",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.59 vs $0.14, 4.4x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-llama-3-3-70b-instruct",
          "aContext": "Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "bf16 · Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Llama 3.3 70B Instruct: price per million tokens by provider (Output)",
          "aValue": 0.79,
          "bValue": 0.4,
          "unit": "usd",
          "aDisplay": "$0.79",
          "bDisplay": "$0.40",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.79 vs $0.40, 2.0x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-llama-3-3-70b-instruct",
          "aContext": "Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "bf16 · Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "groq-vs-sambanova",
      "a": "groq",
      "b": "sambanova",
      "title": "Groq vs SambaNova",
      "seoTitle": "Groq vs SambaNova: price per model",
      "description": "Groq vs SambaNova: 4 list prices per million tokens for the same models, as reported by OpenRouter’s public API, with the gap stated.",
      "verdict": "Groq and SambaNova share 4 reported list prices from one study. No row names a winner: a reported list price has no interval, so each gap is stated, not ranked. Prices change often; the snapshot date is in each row’s context. The rows are 4 unclear; each row says why.",
      "rows": [
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Input)",
          "aValue": 0.15,
          "bValue": 0.14,
          "unit": "usd",
          "aDisplay": "$0.15",
          "bDisplay": "$0.14",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.15 vs $0.14, 1.1x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Output)",
          "aValue": 0.6,
          "bValue": 0.95,
          "unit": "usd",
          "aDisplay": "$0.60",
          "bDisplay": "$0.95",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.60 vs $0.95, 1.6x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Llama 3.3 70B Instruct: price per million tokens by provider (Input)",
          "aValue": 0.59,
          "bValue": 0.45,
          "unit": "usd",
          "aDisplay": "$0.59",
          "bDisplay": "$0.45",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.59 vs $0.45, 1.3x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-llama-3-3-70b-instruct",
          "aContext": "Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Llama 3.3 70B Instruct: price per million tokens by provider (Output)",
          "aValue": 0.79,
          "bValue": 0.9,
          "unit": "usd",
          "aDisplay": "$0.79",
          "bDisplay": "$0.90",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.79 vs $0.90, 1.1x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-llama-3-3-70b-instruct",
          "aContext": "Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "nebius-vs-baseten",
      "a": "nebius",
      "b": "baseten",
      "title": "Nebius vs Baseten",
      "seoTitle": "Nebius vs Baseten: price per model",
      "description": "Nebius vs Baseten: 4 list prices per million tokens for the same models, as reported by OpenRouter’s public API, with the gap stated.",
      "verdict": "Nebius and Baseten share 4 reported list prices from one study. No row names a winner: a reported list price has no interval, so each gap is stated, not ranked. Prices change often; the snapshot date is in each row’s context. The rows are 2 ties and 2 unclear; each row says why.",
      "rows": [
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Input)",
          "aValue": 0.15,
          "bValue": 0.1,
          "unit": "usd",
          "aDisplay": "$0.15",
          "bDisplay": "$0.10",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.15 vs $0.10, 1.5x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Output)",
          "aValue": 0.6,
          "bValue": 0.5,
          "unit": "usd",
          "aDisplay": "$0.60",
          "bDisplay": "$0.50",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.60 vs $0.50, 1.2x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "GLM 5.3: price per million tokens by provider (Input)",
          "aValue": 1.4,
          "bValue": 1.4,
          "unit": "usd",
          "aDisplay": "$1.40",
          "bDisplay": "$1.40",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "aContext": "fp4 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp4 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "GLM 5.3: price per million tokens by provider (Output)",
          "aValue": 4.4,
          "bValue": 4.4,
          "unit": "usd",
          "aDisplay": "$4.40",
          "bDisplay": "$4.40",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "aContext": "fp4 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp4 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "nebius-vs-novita",
      "a": "nebius",
      "b": "novita",
      "title": "Nebius vs Novita AI",
      "seoTitle": "Nebius vs Novita AI: price per model",
      "description": "Nebius vs Novita AI: 4 list prices per million tokens for the same models, as reported by OpenRouter’s public API, with the gap stated.",
      "verdict": "Nebius and Novita AI share 4 reported list prices from one study. No row names a winner: a reported list price has no interval, so each gap is stated, not ranked. Prices change often; the snapshot date is in each row’s context. The rows are 4 unclear; each row says why.",
      "rows": [
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Input)",
          "aValue": 0.15,
          "bValue": 0.05,
          "unit": "usd",
          "aDisplay": "$0.15",
          "bDisplay": "$0.050",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.15 vs $0.050, 3.0x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Output)",
          "aValue": 0.6,
          "bValue": 0.25,
          "unit": "usd",
          "aDisplay": "$0.60",
          "bDisplay": "$0.25",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.60 vs $0.25, 2.4x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "GLM 5.3: price per million tokens by provider (Input)",
          "aValue": 1.4,
          "bValue": 0.42,
          "unit": "usd",
          "aDisplay": "$1.40",
          "bDisplay": "$0.42",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($1.40 vs $0.42, 3.3x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "aContext": "fp4 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "GLM 5.3: price per million tokens by provider (Output)",
          "aValue": 4.4,
          "bValue": 1.32,
          "unit": "usd",
          "aDisplay": "$4.40",
          "bDisplay": "$1.32",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($4.40 vs $1.32, 3.3x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "aContext": "fp4 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "nebius-vs-parasail",
      "a": "nebius",
      "b": "parasail",
      "title": "Nebius vs Parasail",
      "seoTitle": "Nebius vs Parasail: price per model",
      "description": "Nebius vs Parasail: 4 list prices per million tokens for the same models, as reported by OpenRouter’s public API, with the gap stated.",
      "verdict": "Nebius and Parasail share 4 reported list prices from one study. No row names a winner: a reported list price has no interval, so each gap is stated, not ranked. Prices change often; the snapshot date is in each row’s context. The rows are 2 ties and 2 unclear; each row says why.",
      "rows": [
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Input)",
          "aValue": 0.15,
          "bValue": 0.1,
          "unit": "usd",
          "aDisplay": "$0.15",
          "bDisplay": "$0.10",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.15 vs $0.10, 1.5x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Output)",
          "aValue": 0.6,
          "bValue": 0.75,
          "unit": "usd",
          "aDisplay": "$0.60",
          "bDisplay": "$0.75",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.60 vs $0.75, 1.3x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "GLM 5.3: price per million tokens by provider (Input)",
          "aValue": 1.4,
          "bValue": 1.4,
          "unit": "usd",
          "aDisplay": "$1.40",
          "bDisplay": "$1.40",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "aContext": "fp4 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "GLM 5.3: price per million tokens by provider (Output)",
          "aValue": 4.4,
          "bValue": 4.4,
          "unit": "usd",
          "aDisplay": "$4.40",
          "bDisplay": "$4.40",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "aContext": "fp4 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "nebius-vs-siliconflow",
      "a": "nebius",
      "b": "siliconflow",
      "title": "Nebius vs SiliconFlow",
      "seoTitle": "Nebius vs SiliconFlow: price per model",
      "description": "Nebius vs SiliconFlow: 4 list prices per million tokens for the same models, as reported by OpenRouter’s public API, with the gap stated.",
      "verdict": "Nebius and SiliconFlow share 4 reported list prices from one study. No row names a winner: a reported list price has no interval, so each gap is stated, not ranked. Prices change often; the snapshot date is in each row’s context. The rows are 2 ties and 2 unclear; each row says why.",
      "rows": [
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Input)",
          "aValue": 0.15,
          "bValue": 0.15,
          "unit": "usd",
          "aDisplay": "$0.15",
          "bDisplay": "$0.15",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Output)",
          "aValue": 0.6,
          "bValue": 0.6,
          "unit": "usd",
          "aDisplay": "$0.60",
          "bDisplay": "$0.60",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "GLM 5.3: price per million tokens by provider (Input)",
          "aValue": 1.4,
          "bValue": 0.7,
          "unit": "usd",
          "aDisplay": "$1.40",
          "bDisplay": "$0.70",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($1.40 vs $0.70, 2.0x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "aContext": "fp4 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "GLM 5.3: price per million tokens by provider (Output)",
          "aValue": 4.4,
          "bValue": 2.2,
          "unit": "usd",
          "aDisplay": "$4.40",
          "bDisplay": "$2.20",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($4.40 vs $2.20, 2.0x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "aContext": "fp4 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "sambanova-vs-novita",
      "a": "sambanova",
      "b": "novita",
      "title": "SambaNova vs Novita AI",
      "seoTitle": "SambaNova vs Novita AI: price per model",
      "description": "SambaNova vs Novita AI: 4 list prices per million tokens for the same models, as reported by OpenRouter’s public API, with the gap stated.",
      "verdict": "SambaNova and Novita AI share 4 reported list prices from one study. No row names a winner: a reported list price has no interval, so each gap is stated, not ranked. Prices change often; the snapshot date is in each row’s context. The rows are 4 unclear; each row says why.",
      "rows": [
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Input)",
          "aValue": 0.14,
          "bValue": 0.05,
          "unit": "usd",
          "aDisplay": "$0.14",
          "bDisplay": "$0.050",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.14 vs $0.050, 2.8x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Output)",
          "aValue": 0.95,
          "bValue": 0.25,
          "unit": "usd",
          "aDisplay": "$0.95",
          "bDisplay": "$0.25",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.95 vs $0.25, 3.8x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Llama 3.3 70B Instruct: price per million tokens by provider (Input)",
          "aValue": 0.45,
          "bValue": 0.135,
          "unit": "usd",
          "aDisplay": "$0.45",
          "bDisplay": "$0.14",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.45 vs $0.14, 3.3x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-llama-3-3-70b-instruct",
          "aContext": "Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "bf16 · Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Llama 3.3 70B Instruct: price per million tokens by provider (Output)",
          "aValue": 0.9,
          "bValue": 0.4,
          "unit": "usd",
          "aDisplay": "$0.90",
          "bDisplay": "$0.40",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.90 vs $0.40, 2.3x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-llama-3-3-70b-instruct",
          "aContext": "Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "bf16 · Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "sambanova-vs-parasail",
      "a": "sambanova",
      "b": "parasail",
      "title": "SambaNova vs Parasail",
      "seoTitle": "SambaNova vs Parasail: price per model",
      "description": "SambaNova vs Parasail: 4 list prices per million tokens for the same models, as reported by OpenRouter’s public API, with the gap stated.",
      "verdict": "SambaNova and Parasail share 4 reported list prices from one study. No row names a winner: a reported list price has no interval, so each gap is stated, not ranked. Prices change often; the snapshot date is in each row’s context. The rows are 4 unclear; each row says why.",
      "rows": [
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Input)",
          "aValue": 0.14,
          "bValue": 0.1,
          "unit": "usd",
          "aDisplay": "$0.14",
          "bDisplay": "$0.10",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.14 vs $0.10, 1.4x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Output)",
          "aValue": 0.95,
          "bValue": 0.75,
          "unit": "usd",
          "aDisplay": "$0.95",
          "bDisplay": "$0.75",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.95 vs $0.75, 1.3x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Llama 3.3 70B Instruct: price per million tokens by provider (Input)",
          "aValue": 0.45,
          "bValue": 0.22,
          "unit": "usd",
          "aDisplay": "$0.45",
          "bDisplay": "$0.22",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.45 vs $0.22, 2.0x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-llama-3-3-70b-instruct",
          "aContext": "Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Llama 3.3 70B Instruct: price per million tokens by provider (Output)",
          "aValue": 0.9,
          "bValue": 0.5,
          "unit": "usd",
          "aDisplay": "$0.90",
          "bDisplay": "$0.50",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.90 vs $0.50, 1.8x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-llama-3-3-70b-instruct",
          "aContext": "Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "together-vs-nebius",
      "a": "together",
      "b": "nebius",
      "title": "Together AI vs Nebius",
      "seoTitle": "Together AI vs Nebius: price per model",
      "description": "Together AI vs Nebius: 4 list prices per million tokens for the same models, as reported by OpenRouter’s public API, with the gap stated.",
      "verdict": "Together AI and Nebius share 4 reported list prices from one study. No row names a winner: a reported list price has no interval, so each gap is stated, not ranked. Prices change often; the snapshot date is in each row’s context. The rows are 4 ties; each row says why.",
      "rows": [
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Input)",
          "aValue": 0.15,
          "bValue": 0.15,
          "unit": "usd",
          "aDisplay": "$0.15",
          "bDisplay": "$0.15",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Output)",
          "aValue": 0.6,
          "bValue": 0.6,
          "unit": "usd",
          "aDisplay": "$0.60",
          "bDisplay": "$0.60",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "GLM 5.3: price per million tokens by provider (Input)",
          "aValue": 1.4,
          "bValue": 1.4,
          "unit": "usd",
          "aDisplay": "$1.40",
          "bDisplay": "$1.40",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "aContext": "GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp4 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "GLM 5.3: price per million tokens by provider (Output)",
          "aValue": 4.4,
          "bValue": 4.4,
          "unit": "usd",
          "aDisplay": "$4.40",
          "bDisplay": "$4.40",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "aContext": "GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp4 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "together-vs-sambanova",
      "a": "together",
      "b": "sambanova",
      "title": "Together AI vs SambaNova",
      "seoTitle": "Together AI vs SambaNova: price per model",
      "description": "Together AI vs SambaNova: 4 list prices per million tokens for the same models, as reported by OpenRouter’s public API, with the gap stated.",
      "verdict": "Together AI and SambaNova share 4 reported list prices from one study. No row names a winner: a reported list price has no interval, so each gap is stated, not ranked. Prices change often; the snapshot date is in each row’s context. The rows are 4 unclear; each row says why.",
      "rows": [
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Input)",
          "aValue": 0.15,
          "bValue": 0.14,
          "unit": "usd",
          "aDisplay": "$0.15",
          "bDisplay": "$0.14",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.15 vs $0.14, 1.1x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Output)",
          "aValue": 0.6,
          "bValue": 0.95,
          "unit": "usd",
          "aDisplay": "$0.60",
          "bDisplay": "$0.95",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.60 vs $0.95, 1.6x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Llama 3.3 70B Instruct: price per million tokens by provider (Input)",
          "aValue": 1.04,
          "bValue": 0.45,
          "unit": "usd",
          "aDisplay": "$1.04",
          "bDisplay": "$0.45",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($1.04 vs $0.45, 2.3x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-llama-3-3-70b-instruct",
          "aContext": "Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Llama 3.3 70B Instruct: price per million tokens by provider (Output)",
          "aValue": 1.04,
          "bValue": 0.9,
          "unit": "usd",
          "aDisplay": "$1.04",
          "bDisplay": "$0.90",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($1.04 vs $0.90, 1.2x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-llama-3-3-70b-instruct",
          "aContext": "Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "baseten-vs-cloudflare",
      "a": "baseten",
      "b": "cloudflare",
      "title": "Baseten vs Cloudflare Workers AI",
      "seoTitle": "Baseten vs Cloudflare Workers AI: price per model",
      "description": "Baseten vs Cloudflare Workers AI: 3 list prices per million tokens for the same models, as reported by OpenRouter’s public API, with the gap stated.",
      "verdict": "Baseten and Cloudflare Workers AI share 3 reported list prices from one study. No row names a winner: a reported list price has no interval, so each gap is stated, not ranked. Prices change often; the snapshot date is in each row’s context. The rows are 2 ties and 1 unclear; each row says why.",
      "rows": [
        {
          "metric": "GLM 5.3: price per million tokens by provider (Input)",
          "aValue": 1.4,
          "bValue": 1.4,
          "unit": "usd",
          "aDisplay": "$1.40",
          "bDisplay": "$1.40",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "aContext": "fp4 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "GLM 5.3: price per million tokens by provider (Output)",
          "aValue": 4.4,
          "bValue": 4.4,
          "unit": "usd",
          "aDisplay": "$4.40",
          "bDisplay": "$4.40",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "aContext": "fp4 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "GLM 5.3: price per million tokens by provider (Cache read)",
          "aValue": 0.14,
          "bValue": 0.26,
          "unit": "usd",
          "aDisplay": "$0.14",
          "bDisplay": "$0.26",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.14 vs $0.26, 1.9x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "aContext": "fp4 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "cerebras-vs-baseten",
      "a": "cerebras",
      "b": "baseten",
      "title": "Cerebras vs Baseten",
      "seoTitle": "Cerebras vs Baseten: price per model",
      "description": "Cerebras vs Baseten: 3 list prices per million tokens for the same models, as reported by OpenRouter’s public API, with the gap stated.",
      "verdict": "Cerebras and Baseten share 3 reported list prices from one study. No row names a winner: a reported list price has no interval, so each gap is stated, not ranked. Prices change often; the snapshot date is in each row’s context. The rows are 3 unclear; each row says why.",
      "rows": [
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Input)",
          "aValue": 0.35,
          "bValue": 0.1,
          "unit": "usd",
          "aDisplay": "$0.35",
          "bDisplay": "$0.10",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.35 vs $0.10, 3.5x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "fp16 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Output)",
          "aValue": 0.75,
          "bValue": 0.5,
          "unit": "usd",
          "aDisplay": "$0.75",
          "bDisplay": "$0.50",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.75 vs $0.50, 1.5x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "fp16 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Cache read)",
          "aValue": 0.35,
          "bValue": 0.1,
          "unit": "usd",
          "aDisplay": "$0.35",
          "bDisplay": "$0.10",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.35 vs $0.10, 3.5x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "fp16 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "cerebras-vs-parasail",
      "a": "cerebras",
      "b": "parasail",
      "title": "Cerebras vs Parasail",
      "seoTitle": "Cerebras vs Parasail: price per model",
      "description": "Cerebras vs Parasail: 3 list prices per million tokens for the same models, as reported by OpenRouter’s public API, with the gap stated.",
      "verdict": "Cerebras and Parasail share 3 reported list prices from one study. No row names a winner: a reported list price has no interval, so each gap is stated, not ranked. Prices change often; the snapshot date is in each row’s context. The rows are 1 tie and 2 unclear; each row says why.",
      "rows": [
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Input)",
          "aValue": 0.35,
          "bValue": 0.1,
          "unit": "usd",
          "aDisplay": "$0.35",
          "bDisplay": "$0.10",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.35 vs $0.10, 3.5x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "fp16 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Output)",
          "aValue": 0.75,
          "bValue": 0.75,
          "unit": "usd",
          "aDisplay": "$0.75",
          "bDisplay": "$0.75",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "fp16 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Cache read)",
          "aValue": 0.35,
          "bValue": 0.055,
          "unit": "usd",
          "aDisplay": "$0.35",
          "bDisplay": "$0.055",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.35 vs $0.055, 6.4x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "fp16 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "cerebras-vs-siliconflow",
      "a": "cerebras",
      "b": "siliconflow",
      "title": "Cerebras vs SiliconFlow",
      "seoTitle": "Cerebras vs SiliconFlow: price per model",
      "description": "Cerebras vs SiliconFlow: 3 list prices per million tokens for the same models, as reported by OpenRouter’s public API, with the gap stated.",
      "verdict": "Cerebras and SiliconFlow share 3 reported list prices from one study. No row names a winner: a reported list price has no interval, so each gap is stated, not ranked. Prices change often; the snapshot date is in each row’s context. The rows are 3 unclear; each row says why.",
      "rows": [
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Input)",
          "aValue": 0.35,
          "bValue": 0.15,
          "unit": "usd",
          "aDisplay": "$0.35",
          "bDisplay": "$0.15",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.35 vs $0.15, 2.3x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "fp16 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Output)",
          "aValue": 0.75,
          "bValue": 0.6,
          "unit": "usd",
          "aDisplay": "$0.75",
          "bDisplay": "$0.60",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.75 vs $0.60, 1.3x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "fp16 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Cache read)",
          "aValue": 0.35,
          "bValue": 0.075,
          "unit": "usd",
          "aDisplay": "$0.35",
          "bDisplay": "$0.075",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.35 vs $0.075, 4.7x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "fp16 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "claude-haiku-4-5-vs-claude-opus-4-5",
      "a": "claude-haiku-4-5",
      "b": "claude-opus-4-5",
      "title": "Claude Haiku 4.5 vs Claude Opus 4.5",
      "seoTitle": "Claude Haiku 4.5 vs Claude Opus 4.5: measured benchmarks",
      "description": "Claude Haiku 4.5 vs Claude Opus 4.5: 3 measured metrics from 2 studies, with sample sizes, intervals and every failure counted.",
      "verdict": "Claude Haiku 4.5 and Claude Opus 4.5 share 3 measured metrics from 2 studies. No row separates them: every interval or run range overlaps, too few runs were recorded, no interval was recorded, or more is not better for that metric. The rows are 1 tie and 2 unclear; each row says why.",
      "rows": [
        {
          "metric": "Resolved rate on the same 33 SWE-bench Verified instances",
          "aValue": 0.7576,
          "bValue": 0.7273,
          "unit": "rate",
          "aDisplay": "76% (25/33)",
          "bDisplay": "73% (24/33)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Haiku 4.5 59% to 87%; Claude Opus 4.5 56% to 85%), so this sample cannot separate them.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-same-instance-leaderboard",
          "aN": 33,
          "bN": 33,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.5898,
            0.8717
          ],
          "bRange": [
            0.5578,
            0.8493
          ]
        },
        {
          "metric": "Model calls per instance",
          "aValue": 68.5,
          "bValue": 35.9,
          "unit": "calls",
          "aDisplay": "68.5",
          "bDisplay": "35.9",
          "winner": "unclear",
          "basis": "More or fewer calls is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-model-calls",
          "aN": 33,
          "bN": 33,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances"
        },
        {
          "metric": "Recorded cost per resolved instance: Agent vs the public panel",
          "aValue": 0.479,
          "bValue": 1.184,
          "unit": "usd",
          "aDisplay": "$0.48",
          "bDisplay": "$1.18",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.48 vs $1.18, 2.5x) is not tested against run-to-run variation.",
          "studySlug": "cost-thought-experiments",
          "chartId": "cost-per-resolved-agent-vs-panel",
          "aN": 25,
          "bN": 24,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances"
        }
      ]
    },
    {
      "slug": "claude-haiku-4-5-vs-claude-opus-4-6",
      "a": "claude-haiku-4-5",
      "b": "claude-opus-4-6",
      "title": "Claude Haiku 4.5 vs Claude Opus 4.6",
      "seoTitle": "Claude Haiku 4.5 vs Claude Opus 4.6: measured benchmarks",
      "description": "Claude Haiku 4.5 vs Claude Opus 4.6: 3 measured metrics from 2 studies, with sample sizes, intervals and every failure counted.",
      "verdict": "Claude Haiku 4.5 and Claude Opus 4.6 share 3 measured metrics from 2 studies. No row separates them: every interval or run range overlaps, too few runs were recorded, no interval was recorded, or more is not better for that metric. The rows are 1 tie and 2 unclear; each row says why.",
      "rows": [
        {
          "metric": "Resolved rate on the same 33 SWE-bench Verified instances",
          "aValue": 0.7576,
          "bValue": 0.697,
          "unit": "rate",
          "aDisplay": "76% (25/33)",
          "bDisplay": "70% (23/33)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Haiku 4.5 59% to 87%; Claude Opus 4.6 53% to 83%), so this sample cannot separate them.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-same-instance-leaderboard",
          "aN": 33,
          "bN": 33,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "public mini-SWE-agent v2 run, same instances",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.5898,
            0.8717
          ],
          "bRange": [
            0.5266,
            0.8262
          ]
        },
        {
          "metric": "Model calls per instance",
          "aValue": 68.5,
          "bValue": 28.9,
          "unit": "calls",
          "aDisplay": "68.5",
          "bDisplay": "28.9",
          "winner": "unclear",
          "basis": "More or fewer calls is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-model-calls",
          "aN": 33,
          "bN": 33,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "public mini-SWE-agent v2 run, same instances"
        },
        {
          "metric": "Recorded cost per resolved instance: Agent vs the public panel",
          "aValue": 0.479,
          "bValue": 0.875,
          "unit": "usd",
          "aDisplay": "$0.48",
          "bDisplay": "$0.88",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.48 vs $0.88) is not tested against run-to-run variation.",
          "studySlug": "cost-thought-experiments",
          "chartId": "cost-per-resolved-agent-vs-panel",
          "aN": 25,
          "bN": 23,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "public mini-SWE-agent v2 run, same instances"
        }
      ]
    },
    {
      "slug": "claude-haiku-4-5-vs-claude-sonnet-4-5",
      "a": "claude-haiku-4-5",
      "b": "claude-sonnet-4-5",
      "title": "Claude Haiku 4.5 vs Claude Sonnet 4.5",
      "seoTitle": "Claude Haiku 4.5 vs Claude Sonnet 4.5: measured benchmarks",
      "description": "Claude Haiku 4.5 vs Claude Sonnet 4.5: 3 measured metrics from 2 studies, with sample sizes, intervals and every failure counted.",
      "verdict": "Claude Haiku 4.5 and Claude Sonnet 4.5 share 3 measured metrics from 2 studies. No row separates them: every interval or run range overlaps, too few runs were recorded, no interval was recorded, or more is not better for that metric. The rows are 1 tie and 2 unclear; each row says why.",
      "rows": [
        {
          "metric": "Resolved rate on the same 33 SWE-bench Verified instances",
          "aValue": 0.7576,
          "bValue": 0.7576,
          "unit": "rate",
          "aDisplay": "76% (25/33)",
          "bDisplay": "76% (25/33)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Haiku 4.5 59% to 87%; Claude Sonnet 4.5 59% to 87%), so this sample cannot separate them.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-same-instance-leaderboard",
          "aN": 33,
          "bN": 33,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.5898,
            0.8717
          ],
          "bRange": [
            0.5898,
            0.8717
          ]
        },
        {
          "metric": "Model calls per instance",
          "aValue": 68.5,
          "bValue": 51,
          "unit": "calls",
          "aDisplay": "68.5",
          "bDisplay": "51",
          "winner": "unclear",
          "basis": "More or fewer calls is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-model-calls",
          "aN": 33,
          "bN": 33,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances"
        },
        {
          "metric": "Recorded cost per resolved instance: Agent vs the public panel",
          "aValue": 0.479,
          "bValue": 0.913,
          "unit": "usd",
          "aDisplay": "$0.48",
          "bDisplay": "$0.91",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.48 vs $0.91) is not tested against run-to-run variation.",
          "studySlug": "cost-thought-experiments",
          "n": 25,
          "chartId": "cost-per-resolved-agent-vs-panel",
          "aN": 25,
          "bN": 25,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances"
        }
      ]
    },
    {
      "slug": "claude-haiku-4-5-vs-deepseek-v3-2",
      "a": "claude-haiku-4-5",
      "b": "deepseek-v3-2",
      "title": "Claude Haiku 4.5 vs DeepSeek V3.2",
      "seoTitle": "Claude Haiku 4.5 vs DeepSeek V3.2: measured benchmarks",
      "description": "Claude Haiku 4.5 vs DeepSeek V3.2: 3 measured metrics from 2 studies, with sample sizes, intervals and every failure counted.",
      "verdict": "Claude Haiku 4.5 and DeepSeek V3.2 share 3 measured metrics from 2 studies. No row separates them: every interval or run range overlaps, too few runs were recorded, no interval was recorded, or more is not better for that metric. The rows are 1 tie and 2 unclear; each row says why.",
      "rows": [
        {
          "metric": "Resolved rate on the same 33 SWE-bench Verified instances",
          "aValue": 0.7576,
          "bValue": 0.7273,
          "unit": "rate",
          "aDisplay": "76% (25/33)",
          "bDisplay": "73% (24/33)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Haiku 4.5 59% to 87%; DeepSeek V3.2 56% to 85%), so this sample cannot separate them.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-same-instance-leaderboard",
          "aN": 33,
          "bN": 33,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.5898,
            0.8717
          ],
          "bRange": [
            0.5578,
            0.8493
          ]
        },
        {
          "metric": "Model calls per instance",
          "aValue": 68.5,
          "bValue": 88.2,
          "unit": "calls",
          "aDisplay": "68.5",
          "bDisplay": "88.2",
          "winner": "unclear",
          "basis": "More or fewer calls is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-model-calls",
          "aN": 33,
          "bN": 33,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances"
        },
        {
          "metric": "Recorded cost per resolved instance: Agent vs the public panel",
          "aValue": 0.479,
          "bValue": 0.637,
          "unit": "usd",
          "aDisplay": "$0.48",
          "bDisplay": "$0.64",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.48 vs $0.64) is not tested against run-to-run variation.",
          "studySlug": "cost-thought-experiments",
          "chartId": "cost-per-resolved-agent-vs-panel",
          "aN": 25,
          "bN": 24,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances"
        }
      ]
    },
    {
      "slug": "claude-haiku-4-5-vs-gemini-3-flash",
      "a": "claude-haiku-4-5",
      "b": "gemini-3-flash",
      "title": "Claude Haiku 4.5 vs Gemini 3 Flash",
      "seoTitle": "Claude Haiku 4.5 vs Gemini 3 Flash: measured benchmarks",
      "description": "Claude Haiku 4.5 vs Gemini 3 Flash: 3 measured metrics from 2 studies, with sample sizes, intervals and every failure counted.",
      "verdict": "Claude Haiku 4.5 and Gemini 3 Flash share 3 measured metrics from 2 studies. No row separates them: every interval or run range overlaps, too few runs were recorded, no interval was recorded, or more is not better for that metric. The rows are 1 tie and 2 unclear; each row says why.",
      "rows": [
        {
          "metric": "Resolved rate on the same 33 SWE-bench Verified instances",
          "aValue": 0.7576,
          "bValue": 0.8182,
          "unit": "rate",
          "aDisplay": "76% (25/33)",
          "bDisplay": "82% (27/33)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Haiku 4.5 59% to 87%; Gemini 3 Flash 66% to 91%), so this sample cannot separate them.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-same-instance-leaderboard",
          "aN": 33,
          "bN": 33,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.5898,
            0.8717
          ],
          "bRange": [
            0.6561,
            0.9139
          ]
        },
        {
          "metric": "Model calls per instance",
          "aValue": 68.5,
          "bValue": 54.2,
          "unit": "calls",
          "aDisplay": "68.5",
          "bDisplay": "54.2",
          "winner": "unclear",
          "basis": "More or fewer calls is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-model-calls",
          "aN": 33,
          "bN": 33,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances"
        },
        {
          "metric": "Recorded cost per resolved instance: Agent vs the public panel",
          "aValue": 0.479,
          "bValue": 0.436,
          "unit": "usd",
          "aDisplay": "$0.48",
          "bDisplay": "$0.44",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.48 vs $0.44) is not tested against run-to-run variation.",
          "studySlug": "cost-thought-experiments",
          "chartId": "cost-per-resolved-agent-vs-panel",
          "aN": 25,
          "bN": 27,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances"
        }
      ]
    },
    {
      "slug": "claude-haiku-4-5-vs-glm-5",
      "a": "claude-haiku-4-5",
      "b": "glm-5",
      "title": "Claude Haiku 4.5 vs GLM 5",
      "seoTitle": "Claude Haiku 4.5 vs GLM 5: measured benchmarks",
      "description": "Claude Haiku 4.5 vs GLM 5: 3 measured metrics from 2 studies (Resolved rate on the same 33 SWE-bench Verified instances; more), with sample sizes and intervals.",
      "verdict": "Claude Haiku 4.5 and GLM 5 share 3 measured metrics from 2 studies. No row separates them: every interval or run range overlaps, too few runs were recorded, no interval was recorded, or more is not better for that metric. The rows are 1 tie and 2 unclear; each row says why.",
      "rows": [
        {
          "metric": "Resolved rate on the same 33 SWE-bench Verified instances",
          "aValue": 0.7576,
          "bValue": 0.7879,
          "unit": "rate",
          "aDisplay": "76% (25/33)",
          "bDisplay": "79% (26/33)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Haiku 4.5 59% to 87%; GLM 5 62% to 89%), so this sample cannot separate them.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-same-instance-leaderboard",
          "aN": 33,
          "bN": 33,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.5898,
            0.8717
          ],
          "bRange": [
            0.6225,
            0.8932
          ]
        },
        {
          "metric": "Model calls per instance",
          "aValue": 68.5,
          "bValue": 77.5,
          "unit": "calls",
          "aDisplay": "68.5",
          "bDisplay": "77.5",
          "winner": "unclear",
          "basis": "More or fewer calls is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-model-calls",
          "aN": 33,
          "bN": 33,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances"
        },
        {
          "metric": "Recorded cost per resolved instance: Agent vs the public panel",
          "aValue": 0.479,
          "bValue": 0.667,
          "unit": "usd",
          "aDisplay": "$0.48",
          "bDisplay": "$0.67",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.48 vs $0.67) is not tested against run-to-run variation.",
          "studySlug": "cost-thought-experiments",
          "chartId": "cost-per-resolved-agent-vs-panel",
          "aN": 25,
          "bN": 26,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances"
        }
      ]
    },
    {
      "slug": "claude-haiku-4-5-vs-gpt-5-2",
      "a": "claude-haiku-4-5",
      "b": "gpt-5-2",
      "title": "Claude Haiku 4.5 vs GPT 5.2",
      "seoTitle": "Claude Haiku 4.5 vs GPT 5.2: measured benchmarks",
      "description": "Claude Haiku 4.5 vs GPT 5.2: 3 measured metrics from 2 studies, with sample sizes, intervals and every failure counted.",
      "verdict": "Claude Haiku 4.5 and GPT 5.2 share 3 measured metrics from 2 studies. No row separates them: every interval or run range overlaps, too few runs were recorded, no interval was recorded, or more is not better for that metric. The rows are 1 tie and 2 unclear; each row says why.",
      "rows": [
        {
          "metric": "Resolved rate on the same 33 SWE-bench Verified instances",
          "aValue": 0.7576,
          "bValue": 0.8485,
          "unit": "rate",
          "aDisplay": "76% (25/33)",
          "bDisplay": "85% (28/33)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Haiku 4.5 59% to 87%; GPT 5.2 69% to 93%), so this sample cannot separate them.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-same-instance-leaderboard",
          "aN": 33,
          "bN": 33,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.5898,
            0.8717
          ],
          "bRange": [
            0.6908,
            0.9335
          ]
        },
        {
          "metric": "Model calls per instance",
          "aValue": 68.5,
          "bValue": 35.6,
          "unit": "calls",
          "aDisplay": "68.5",
          "bDisplay": "35.6",
          "winner": "unclear",
          "basis": "More or fewer calls is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-model-calls",
          "aN": 33,
          "bN": 33,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances"
        },
        {
          "metric": "Recorded cost per resolved instance: Agent vs the public panel",
          "aValue": 0.479,
          "bValue": 0.628,
          "unit": "usd",
          "aDisplay": "$0.48",
          "bDisplay": "$0.63",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.48 vs $0.63) is not tested against run-to-run variation.",
          "studySlug": "cost-thought-experiments",
          "chartId": "cost-per-resolved-agent-vs-panel",
          "aN": 25,
          "bN": 28,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances"
        }
      ]
    },
    {
      "slug": "claude-haiku-4-5-vs-gpt-5-mini",
      "a": "claude-haiku-4-5",
      "b": "gpt-5-mini",
      "title": "Claude Haiku 4.5 vs GPT 5 mini",
      "seoTitle": "Claude Haiku 4.5 vs GPT 5 mini: measured benchmarks",
      "description": "Claude Haiku 4.5 vs GPT 5 mini: 3 measured metrics from 2 studies, with sample sizes, intervals and every failure counted.",
      "verdict": "Claude Haiku 4.5 and GPT 5 mini share 3 measured metrics from 2 studies. No row separates them: every interval or run range overlaps, too few runs were recorded, no interval was recorded, or more is not better for that metric. The rows are 1 tie and 2 unclear; each row says why.",
      "rows": [
        {
          "metric": "Resolved rate on the same 33 SWE-bench Verified instances",
          "aValue": 0.7576,
          "bValue": 0.6364,
          "unit": "rate",
          "aDisplay": "76% (25/33)",
          "bDisplay": "64% (21/33)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Haiku 4.5 59% to 87%; GPT 5 mini 47% to 78%), so this sample cannot separate them.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-same-instance-leaderboard",
          "aN": 33,
          "bN": 33,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "public mini-SWE-agent v2 run, same instances",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.5898,
            0.8717
          ],
          "bRange": [
            0.4662,
            0.7781
          ]
        },
        {
          "metric": "Model calls per instance",
          "aValue": 68.5,
          "bValue": 20.8,
          "unit": "calls",
          "aDisplay": "68.5",
          "bDisplay": "20.8",
          "winner": "unclear",
          "basis": "More or fewer calls is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-model-calls",
          "aN": 33,
          "bN": 33,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "public mini-SWE-agent v2 run, same instances"
        },
        {
          "metric": "Recorded cost per resolved instance: Agent vs the public panel",
          "aValue": 0.479,
          "bValue": 0.08,
          "unit": "usd",
          "aDisplay": "$0.48",
          "bDisplay": "$0.080",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.48 vs $0.080, 6.0x) is not tested against run-to-run variation.",
          "studySlug": "cost-thought-experiments",
          "chartId": "cost-per-resolved-agent-vs-panel",
          "aN": 25,
          "bN": 21,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "public mini-SWE-agent v2 run, same instances"
        }
      ]
    },
    {
      "slug": "claude-haiku-4-5-vs-kimi-k2-5",
      "a": "claude-haiku-4-5",
      "b": "kimi-k2-5",
      "title": "Claude Haiku 4.5 vs Kimi K2.5",
      "seoTitle": "Claude Haiku 4.5 vs Kimi K2.5: measured benchmarks",
      "description": "Claude Haiku 4.5 vs Kimi K2.5: 3 measured metrics from 2 studies, with sample sizes, intervals and every failure counted.",
      "verdict": "Claude Haiku 4.5 and Kimi K2.5 share 3 measured metrics from 2 studies. No row separates them: every interval or run range overlaps, too few runs were recorded, no interval was recorded, or more is not better for that metric. The rows are 1 tie and 2 unclear; each row says why.",
      "rows": [
        {
          "metric": "Resolved rate on the same 33 SWE-bench Verified instances",
          "aValue": 0.7576,
          "bValue": 0.697,
          "unit": "rate",
          "aDisplay": "76% (25/33)",
          "bDisplay": "70% (23/33)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Haiku 4.5 59% to 87%; Kimi K2.5 53% to 83%), so this sample cannot separate them.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-same-instance-leaderboard",
          "aN": 33,
          "bN": 33,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.5898,
            0.8717
          ],
          "bRange": [
            0.5266,
            0.8262
          ]
        },
        {
          "metric": "Model calls per instance",
          "aValue": 68.5,
          "bValue": 56.7,
          "unit": "calls",
          "aDisplay": "68.5",
          "bDisplay": "56.7",
          "winner": "unclear",
          "basis": "More or fewer calls is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-model-calls",
          "aN": 33,
          "bN": 33,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances"
        },
        {
          "metric": "Recorded cost per resolved instance: Agent vs the public panel",
          "aValue": 0.479,
          "bValue": 0.256,
          "unit": "usd",
          "aDisplay": "$0.48",
          "bDisplay": "$0.26",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.48 vs $0.26) is not tested against run-to-run variation.",
          "studySlug": "cost-thought-experiments",
          "chartId": "cost-per-resolved-agent-vs-panel",
          "aN": 25,
          "bN": 23,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances"
        }
      ]
    },
    {
      "slug": "claude-haiku-4-5-vs-minimax-m2-5",
      "a": "claude-haiku-4-5",
      "b": "minimax-m2-5",
      "title": "Claude Haiku 4.5 vs MiniMax M2.5",
      "seoTitle": "Claude Haiku 4.5 vs MiniMax M2.5: measured benchmarks",
      "description": "Claude Haiku 4.5 vs MiniMax M2.5: 3 measured metrics from 2 studies, with sample sizes, intervals and every failure counted.",
      "verdict": "Claude Haiku 4.5 and MiniMax M2.5 share 3 measured metrics from 2 studies. No row separates them: every interval or run range overlaps, too few runs were recorded, no interval was recorded, or more is not better for that metric. The rows are 1 tie and 2 unclear; each row says why.",
      "rows": [
        {
          "metric": "Resolved rate on the same 33 SWE-bench Verified instances",
          "aValue": 0.7576,
          "bValue": 0.697,
          "unit": "rate",
          "aDisplay": "76% (25/33)",
          "bDisplay": "70% (23/33)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Haiku 4.5 59% to 87%; MiniMax M2.5 53% to 83%), so this sample cannot separate them.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-same-instance-leaderboard",
          "aN": 33,
          "bN": 33,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.5898,
            0.8717
          ],
          "bRange": [
            0.5266,
            0.8262
          ]
        },
        {
          "metric": "Model calls per instance",
          "aValue": 68.5,
          "bValue": 58.4,
          "unit": "calls",
          "aDisplay": "68.5",
          "bDisplay": "58.4",
          "winner": "unclear",
          "basis": "More or fewer calls is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-model-calls",
          "aN": 33,
          "bN": 33,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances"
        },
        {
          "metric": "Recorded cost per resolved instance: Agent vs the public panel",
          "aValue": 0.479,
          "bValue": 0.107,
          "unit": "usd",
          "aDisplay": "$0.48",
          "bDisplay": "$0.11",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.48 vs $0.11, 4.5x) is not tested against run-to-run variation.",
          "studySlug": "cost-thought-experiments",
          "chartId": "cost-per-resolved-agent-vs-panel",
          "aN": 25,
          "bN": 23,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances"
        }
      ]
    },
    {
      "slug": "claude-opus-4-5-vs-claude-opus-4-6",
      "a": "claude-opus-4-5",
      "b": "claude-opus-4-6",
      "title": "Claude Opus 4.5 vs Claude Opus 4.6",
      "seoTitle": "Claude Opus 4.5 vs Claude Opus 4.6: measured benchmarks",
      "description": "Claude Opus 4.5 vs Claude Opus 4.6: 3 measured metrics from 2 studies, with sample sizes, intervals and every failure counted.",
      "verdict": "Claude Opus 4.5 and Claude Opus 4.6 share 3 measured metrics from 2 studies. No row separates them: every interval or run range overlaps, too few runs were recorded, no interval was recorded, or more is not better for that metric. The rows are 1 tie and 2 unclear; each row says why.",
      "rows": [
        {
          "metric": "Resolved rate on the same 33 SWE-bench Verified instances",
          "aValue": 0.7273,
          "bValue": 0.697,
          "unit": "rate",
          "aDisplay": "73% (24/33)",
          "bDisplay": "70% (23/33)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Opus 4.5 56% to 85%; Claude Opus 4.6 53% to 83%), so this sample cannot separate them.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-same-instance-leaderboard",
          "aN": 33,
          "bN": 33,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "public mini-SWE-agent v2 run, same instances",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.5578,
            0.8493
          ],
          "bRange": [
            0.5266,
            0.8262
          ]
        },
        {
          "metric": "Model calls per instance",
          "aValue": 35.9,
          "bValue": 28.9,
          "unit": "calls",
          "aDisplay": "35.9",
          "bDisplay": "28.9",
          "winner": "unclear",
          "basis": "More or fewer calls is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-model-calls",
          "aN": 33,
          "bN": 33,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "public mini-SWE-agent v2 run, same instances"
        },
        {
          "metric": "Recorded cost per resolved instance: Agent vs the public panel",
          "aValue": 1.184,
          "bValue": 0.875,
          "unit": "usd",
          "aDisplay": "$1.18",
          "bDisplay": "$0.88",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($1.18 vs $0.88) is not tested against run-to-run variation.",
          "studySlug": "cost-thought-experiments",
          "chartId": "cost-per-resolved-agent-vs-panel",
          "aN": 24,
          "bN": 23,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "public mini-SWE-agent v2 run, same instances"
        }
      ]
    },
    {
      "slug": "claude-opus-4-5-vs-deepseek-v3-2",
      "a": "claude-opus-4-5",
      "b": "deepseek-v3-2",
      "title": "Claude Opus 4.5 vs DeepSeek V3.2",
      "seoTitle": "Claude Opus 4.5 vs DeepSeek V3.2: measured benchmarks",
      "description": "Claude Opus 4.5 vs DeepSeek V3.2: 3 measured metrics from 2 studies, with sample sizes, intervals and every failure counted.",
      "verdict": "Claude Opus 4.5 and DeepSeek V3.2 share 3 measured metrics from 2 studies. No row separates them: every interval or run range overlaps, too few runs were recorded, no interval was recorded, or more is not better for that metric. The rows are 1 tie and 2 unclear; each row says why.",
      "rows": [
        {
          "metric": "Resolved rate on the same 33 SWE-bench Verified instances",
          "aValue": 0.7273,
          "bValue": 0.7273,
          "unit": "rate",
          "aDisplay": "73% (24/33)",
          "bDisplay": "73% (24/33)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Opus 4.5 56% to 85%; DeepSeek V3.2 56% to 85%), so this sample cannot separate them.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-same-instance-leaderboard",
          "aN": 33,
          "bN": 33,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.5578,
            0.8493
          ],
          "bRange": [
            0.5578,
            0.8493
          ]
        },
        {
          "metric": "Model calls per instance",
          "aValue": 35.9,
          "bValue": 88.2,
          "unit": "calls",
          "aDisplay": "35.9",
          "bDisplay": "88.2",
          "winner": "unclear",
          "basis": "More or fewer calls is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-model-calls",
          "aN": 33,
          "bN": 33,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances"
        },
        {
          "metric": "Recorded cost per resolved instance: Agent vs the public panel",
          "aValue": 1.184,
          "bValue": 0.637,
          "unit": "usd",
          "aDisplay": "$1.18",
          "bDisplay": "$0.64",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($1.18 vs $0.64) is not tested against run-to-run variation.",
          "studySlug": "cost-thought-experiments",
          "n": 24,
          "chartId": "cost-per-resolved-agent-vs-panel",
          "aN": 24,
          "bN": 24,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances"
        }
      ]
    },
    {
      "slug": "claude-opus-4-5-vs-gpt-5-mini",
      "a": "claude-opus-4-5",
      "b": "gpt-5-mini",
      "title": "Claude Opus 4.5 vs GPT 5 mini",
      "seoTitle": "Claude Opus 4.5 vs GPT 5 mini: measured benchmarks",
      "description": "Claude Opus 4.5 vs GPT 5 mini: 3 measured metrics from 2 studies, with sample sizes, intervals and every failure counted.",
      "verdict": "Claude Opus 4.5 and GPT 5 mini share 3 measured metrics from 2 studies. No row separates them: every interval or run range overlaps, too few runs were recorded, no interval was recorded, or more is not better for that metric. The rows are 1 tie and 2 unclear; each row says why.",
      "rows": [
        {
          "metric": "Resolved rate on the same 33 SWE-bench Verified instances",
          "aValue": 0.7273,
          "bValue": 0.6364,
          "unit": "rate",
          "aDisplay": "73% (24/33)",
          "bDisplay": "64% (21/33)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Opus 4.5 56% to 85%; GPT 5 mini 47% to 78%), so this sample cannot separate them.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-same-instance-leaderboard",
          "aN": 33,
          "bN": 33,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "public mini-SWE-agent v2 run, same instances",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.5578,
            0.8493
          ],
          "bRange": [
            0.4662,
            0.7781
          ]
        },
        {
          "metric": "Model calls per instance",
          "aValue": 35.9,
          "bValue": 20.8,
          "unit": "calls",
          "aDisplay": "35.9",
          "bDisplay": "20.8",
          "winner": "unclear",
          "basis": "More or fewer calls is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-model-calls",
          "aN": 33,
          "bN": 33,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "public mini-SWE-agent v2 run, same instances"
        },
        {
          "metric": "Recorded cost per resolved instance: Agent vs the public panel",
          "aValue": 1.184,
          "bValue": 0.08,
          "unit": "usd",
          "aDisplay": "$1.18",
          "bDisplay": "$0.080",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($1.18 vs $0.080, 15x) is not tested against run-to-run variation.",
          "studySlug": "cost-thought-experiments",
          "chartId": "cost-per-resolved-agent-vs-panel",
          "aN": 24,
          "bN": 21,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "public mini-SWE-agent v2 run, same instances"
        }
      ]
    },
    {
      "slug": "claude-opus-4-5-vs-kimi-k2-5",
      "a": "claude-opus-4-5",
      "b": "kimi-k2-5",
      "title": "Claude Opus 4.5 vs Kimi K2.5",
      "seoTitle": "Claude Opus 4.5 vs Kimi K2.5: measured benchmarks",
      "description": "Claude Opus 4.5 vs Kimi K2.5: 3 measured metrics from 2 studies, with sample sizes, intervals and every failure counted.",
      "verdict": "Claude Opus 4.5 and Kimi K2.5 share 3 measured metrics from 2 studies. No row separates them: every interval or run range overlaps, too few runs were recorded, no interval was recorded, or more is not better for that metric. The rows are 1 tie and 2 unclear; each row says why.",
      "rows": [
        {
          "metric": "Resolved rate on the same 33 SWE-bench Verified instances",
          "aValue": 0.7273,
          "bValue": 0.697,
          "unit": "rate",
          "aDisplay": "73% (24/33)",
          "bDisplay": "70% (23/33)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Opus 4.5 56% to 85%; Kimi K2.5 53% to 83%), so this sample cannot separate them.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-same-instance-leaderboard",
          "aN": 33,
          "bN": 33,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.5578,
            0.8493
          ],
          "bRange": [
            0.5266,
            0.8262
          ]
        },
        {
          "metric": "Model calls per instance",
          "aValue": 35.9,
          "bValue": 56.7,
          "unit": "calls",
          "aDisplay": "35.9",
          "bDisplay": "56.7",
          "winner": "unclear",
          "basis": "More or fewer calls is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-model-calls",
          "aN": 33,
          "bN": 33,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances"
        },
        {
          "metric": "Recorded cost per resolved instance: Agent vs the public panel",
          "aValue": 1.184,
          "bValue": 0.256,
          "unit": "usd",
          "aDisplay": "$1.18",
          "bDisplay": "$0.26",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($1.18 vs $0.26, 4.6x) is not tested against run-to-run variation.",
          "studySlug": "cost-thought-experiments",
          "chartId": "cost-per-resolved-agent-vs-panel",
          "aN": 24,
          "bN": 23,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances"
        }
      ]
    },
    {
      "slug": "claude-opus-4-5-vs-minimax-m2-5",
      "a": "claude-opus-4-5",
      "b": "minimax-m2-5",
      "title": "Claude Opus 4.5 vs MiniMax M2.5",
      "seoTitle": "Claude Opus 4.5 vs MiniMax M2.5: measured benchmarks",
      "description": "Claude Opus 4.5 vs MiniMax M2.5: 3 measured metrics from 2 studies, with sample sizes, intervals and every failure counted.",
      "verdict": "Claude Opus 4.5 and MiniMax M2.5 share 3 measured metrics from 2 studies. No row separates them: every interval or run range overlaps, too few runs were recorded, no interval was recorded, or more is not better for that metric. The rows are 1 tie and 2 unclear; each row says why.",
      "rows": [
        {
          "metric": "Resolved rate on the same 33 SWE-bench Verified instances",
          "aValue": 0.7273,
          "bValue": 0.697,
          "unit": "rate",
          "aDisplay": "73% (24/33)",
          "bDisplay": "70% (23/33)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Opus 4.5 56% to 85%; MiniMax M2.5 53% to 83%), so this sample cannot separate them.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-same-instance-leaderboard",
          "aN": 33,
          "bN": 33,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.5578,
            0.8493
          ],
          "bRange": [
            0.5266,
            0.8262
          ]
        },
        {
          "metric": "Model calls per instance",
          "aValue": 35.9,
          "bValue": 58.4,
          "unit": "calls",
          "aDisplay": "35.9",
          "bDisplay": "58.4",
          "winner": "unclear",
          "basis": "More or fewer calls is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-model-calls",
          "aN": 33,
          "bN": 33,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances"
        },
        {
          "metric": "Recorded cost per resolved instance: Agent vs the public panel",
          "aValue": 1.184,
          "bValue": 0.107,
          "unit": "usd",
          "aDisplay": "$1.18",
          "bDisplay": "$0.11",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($1.18 vs $0.11, 11x) is not tested against run-to-run variation.",
          "studySlug": "cost-thought-experiments",
          "chartId": "cost-per-resolved-agent-vs-panel",
          "aN": 24,
          "bN": 23,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances"
        }
      ]
    },
    {
      "slug": "claude-opus-4-6-vs-deepseek-v3-2",
      "a": "claude-opus-4-6",
      "b": "deepseek-v3-2",
      "title": "Claude Opus 4.6 vs DeepSeek V3.2",
      "seoTitle": "Claude Opus 4.6 vs DeepSeek V3.2: measured benchmarks",
      "description": "Claude Opus 4.6 vs DeepSeek V3.2: 3 measured metrics from 2 studies, with sample sizes, intervals and every failure counted.",
      "verdict": "Claude Opus 4.6 and DeepSeek V3.2 share 3 measured metrics from 2 studies. No row separates them: every interval or run range overlaps, too few runs were recorded, no interval was recorded, or more is not better for that metric. The rows are 1 tie and 2 unclear; each row says why.",
      "rows": [
        {
          "metric": "Resolved rate on the same 33 SWE-bench Verified instances",
          "aValue": 0.697,
          "bValue": 0.7273,
          "unit": "rate",
          "aDisplay": "70% (23/33)",
          "bDisplay": "73% (24/33)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Opus 4.6 53% to 83%; DeepSeek V3.2 56% to 85%), so this sample cannot separate them.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-same-instance-leaderboard",
          "aN": 33,
          "bN": 33,
          "aContext": "public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.5266,
            0.8262
          ],
          "bRange": [
            0.5578,
            0.8493
          ]
        },
        {
          "metric": "Model calls per instance",
          "aValue": 28.9,
          "bValue": 88.2,
          "unit": "calls",
          "aDisplay": "28.9",
          "bDisplay": "88.2",
          "winner": "unclear",
          "basis": "More or fewer calls is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-model-calls",
          "aN": 33,
          "bN": 33,
          "aContext": "public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances"
        },
        {
          "metric": "Recorded cost per resolved instance: Agent vs the public panel",
          "aValue": 0.875,
          "bValue": 0.637,
          "unit": "usd",
          "aDisplay": "$0.88",
          "bDisplay": "$0.64",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.88 vs $0.64) is not tested against run-to-run variation.",
          "studySlug": "cost-thought-experiments",
          "chartId": "cost-per-resolved-agent-vs-panel",
          "aN": 23,
          "bN": 24,
          "aContext": "public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances"
        }
      ]
    },
    {
      "slug": "claude-opus-4-6-vs-gpt-5-mini",
      "a": "claude-opus-4-6",
      "b": "gpt-5-mini",
      "title": "Claude Opus 4.6 vs GPT 5 mini",
      "seoTitle": "Claude Opus 4.6 vs GPT 5 mini: measured benchmarks",
      "description": "Claude Opus 4.6 vs GPT 5 mini: 3 measured metrics from 2 studies, with sample sizes, intervals and every failure counted.",
      "verdict": "Claude Opus 4.6 and GPT 5 mini share 3 measured metrics from 2 studies. No row separates them: every interval or run range overlaps, too few runs were recorded, no interval was recorded, or more is not better for that metric. The rows are 1 tie and 2 unclear; each row says why.",
      "rows": [
        {
          "metric": "Resolved rate on the same 33 SWE-bench Verified instances",
          "aValue": 0.697,
          "bValue": 0.6364,
          "unit": "rate",
          "aDisplay": "70% (23/33)",
          "bDisplay": "64% (21/33)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Opus 4.6 53% to 83%; GPT 5 mini 47% to 78%), so this sample cannot separate them.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-same-instance-leaderboard",
          "aN": 33,
          "bN": 33,
          "aContext": "public mini-SWE-agent v2 run, same instances",
          "bContext": "public mini-SWE-agent v2 run, same instances",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.5266,
            0.8262
          ],
          "bRange": [
            0.4662,
            0.7781
          ]
        },
        {
          "metric": "Model calls per instance",
          "aValue": 28.9,
          "bValue": 20.8,
          "unit": "calls",
          "aDisplay": "28.9",
          "bDisplay": "20.8",
          "winner": "unclear",
          "basis": "More or fewer calls is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-model-calls",
          "aN": 33,
          "bN": 33,
          "aContext": "public mini-SWE-agent v2 run, same instances",
          "bContext": "public mini-SWE-agent v2 run, same instances"
        },
        {
          "metric": "Recorded cost per resolved instance: Agent vs the public panel",
          "aValue": 0.875,
          "bValue": 0.08,
          "unit": "usd",
          "aDisplay": "$0.88",
          "bDisplay": "$0.080",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.88 vs $0.080, 11x) is not tested against run-to-run variation.",
          "studySlug": "cost-thought-experiments",
          "chartId": "cost-per-resolved-agent-vs-panel",
          "aN": 23,
          "bN": 21,
          "aContext": "public mini-SWE-agent v2 run, same instances",
          "bContext": "public mini-SWE-agent v2 run, same instances"
        }
      ]
    },
    {
      "slug": "claude-opus-4-6-vs-kimi-k2-5",
      "a": "claude-opus-4-6",
      "b": "kimi-k2-5",
      "title": "Claude Opus 4.6 vs Kimi K2.5",
      "seoTitle": "Claude Opus 4.6 vs Kimi K2.5: measured benchmarks",
      "description": "Claude Opus 4.6 vs Kimi K2.5: 3 measured metrics from 2 studies, with sample sizes, intervals and every failure counted.",
      "verdict": "Claude Opus 4.6 and Kimi K2.5 share 3 measured metrics from 2 studies. No row separates them: every interval or run range overlaps, too few runs were recorded, no interval was recorded, or more is not better for that metric. The rows are 1 tie and 2 unclear; each row says why.",
      "rows": [
        {
          "metric": "Resolved rate on the same 33 SWE-bench Verified instances",
          "aValue": 0.697,
          "bValue": 0.697,
          "unit": "rate",
          "aDisplay": "70% (23/33)",
          "bDisplay": "70% (23/33)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Opus 4.6 53% to 83%; Kimi K2.5 53% to 83%), so this sample cannot separate them.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-same-instance-leaderboard",
          "aN": 33,
          "bN": 33,
          "aContext": "public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.5266,
            0.8262
          ],
          "bRange": [
            0.5266,
            0.8262
          ]
        },
        {
          "metric": "Model calls per instance",
          "aValue": 28.9,
          "bValue": 56.7,
          "unit": "calls",
          "aDisplay": "28.9",
          "bDisplay": "56.7",
          "winner": "unclear",
          "basis": "More or fewer calls is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-model-calls",
          "aN": 33,
          "bN": 33,
          "aContext": "public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances"
        },
        {
          "metric": "Recorded cost per resolved instance: Agent vs the public panel",
          "aValue": 0.875,
          "bValue": 0.256,
          "unit": "usd",
          "aDisplay": "$0.88",
          "bDisplay": "$0.26",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.88 vs $0.26, 3.4x) is not tested against run-to-run variation.",
          "studySlug": "cost-thought-experiments",
          "n": 23,
          "chartId": "cost-per-resolved-agent-vs-panel",
          "aN": 23,
          "bN": 23,
          "aContext": "public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances"
        }
      ]
    },
    {
      "slug": "claude-opus-4-6-vs-minimax-m2-5",
      "a": "claude-opus-4-6",
      "b": "minimax-m2-5",
      "title": "Claude Opus 4.6 vs MiniMax M2.5",
      "seoTitle": "Claude Opus 4.6 vs MiniMax M2.5: measured benchmarks",
      "description": "Claude Opus 4.6 vs MiniMax M2.5: 3 measured metrics from 2 studies, with sample sizes, intervals and every failure counted.",
      "verdict": "Claude Opus 4.6 and MiniMax M2.5 share 3 measured metrics from 2 studies. No row separates them: every interval or run range overlaps, too few runs were recorded, no interval was recorded, or more is not better for that metric. The rows are 1 tie and 2 unclear; each row says why.",
      "rows": [
        {
          "metric": "Resolved rate on the same 33 SWE-bench Verified instances",
          "aValue": 0.697,
          "bValue": 0.697,
          "unit": "rate",
          "aDisplay": "70% (23/33)",
          "bDisplay": "70% (23/33)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Opus 4.6 53% to 83%; MiniMax M2.5 53% to 83%), so this sample cannot separate them.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-same-instance-leaderboard",
          "aN": 33,
          "bN": 33,
          "aContext": "public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.5266,
            0.8262
          ],
          "bRange": [
            0.5266,
            0.8262
          ]
        },
        {
          "metric": "Model calls per instance",
          "aValue": 28.9,
          "bValue": 58.4,
          "unit": "calls",
          "aDisplay": "28.9",
          "bDisplay": "58.4",
          "winner": "unclear",
          "basis": "More or fewer calls is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-model-calls",
          "aN": 33,
          "bN": 33,
          "aContext": "public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances"
        },
        {
          "metric": "Recorded cost per resolved instance: Agent vs the public panel",
          "aValue": 0.875,
          "bValue": 0.107,
          "unit": "usd",
          "aDisplay": "$0.88",
          "bDisplay": "$0.11",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.88 vs $0.11, 8.2x) is not tested against run-to-run variation.",
          "studySlug": "cost-thought-experiments",
          "n": 23,
          "chartId": "cost-per-resolved-agent-vs-panel",
          "aN": 23,
          "bN": 23,
          "aContext": "public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances"
        }
      ]
    },
    {
      "slug": "claude-sonnet-4-5-vs-claude-opus-4-5",
      "a": "claude-sonnet-4-5",
      "b": "claude-opus-4-5",
      "title": "Claude Sonnet 4.5 vs Claude Opus 4.5",
      "seoTitle": "Claude Sonnet 4.5 vs Claude Opus 4.5: measured benchmarks",
      "description": "Claude Sonnet 4.5 vs Claude Opus 4.5: 3 measured metrics from 2 studies, with sample sizes, intervals and every failure counted.",
      "verdict": "Claude Sonnet 4.5 and Claude Opus 4.5 share 3 measured metrics from 2 studies. No row separates them: every interval or run range overlaps, too few runs were recorded, no interval was recorded, or more is not better for that metric. The rows are 1 tie and 2 unclear; each row says why.",
      "rows": [
        {
          "metric": "Resolved rate on the same 33 SWE-bench Verified instances",
          "aValue": 0.7576,
          "bValue": 0.7273,
          "unit": "rate",
          "aDisplay": "76% (25/33)",
          "bDisplay": "73% (24/33)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Sonnet 4.5 59% to 87%; Claude Opus 4.5 56% to 85%), so this sample cannot separate them.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-same-instance-leaderboard",
          "aN": 33,
          "bN": 33,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.5898,
            0.8717
          ],
          "bRange": [
            0.5578,
            0.8493
          ]
        },
        {
          "metric": "Model calls per instance",
          "aValue": 51,
          "bValue": 35.9,
          "unit": "calls",
          "aDisplay": "51",
          "bDisplay": "35.9",
          "winner": "unclear",
          "basis": "More or fewer calls is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-model-calls",
          "aN": 33,
          "bN": 33,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances"
        },
        {
          "metric": "Recorded cost per resolved instance: Agent vs the public panel",
          "aValue": 0.913,
          "bValue": 1.184,
          "unit": "usd",
          "aDisplay": "$0.91",
          "bDisplay": "$1.18",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.91 vs $1.18) is not tested against run-to-run variation.",
          "studySlug": "cost-thought-experiments",
          "chartId": "cost-per-resolved-agent-vs-panel",
          "aN": 25,
          "bN": 24,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances"
        }
      ]
    },
    {
      "slug": "claude-sonnet-4-5-vs-claude-opus-4-6",
      "a": "claude-sonnet-4-5",
      "b": "claude-opus-4-6",
      "title": "Claude Sonnet 4.5 vs Claude Opus 4.6",
      "seoTitle": "Claude Sonnet 4.5 vs Claude Opus 4.6: measured benchmarks",
      "description": "Claude Sonnet 4.5 vs Claude Opus 4.6: 3 measured metrics from 2 studies, with sample sizes, intervals and every failure counted.",
      "verdict": "Claude Sonnet 4.5 and Claude Opus 4.6 share 3 measured metrics from 2 studies. No row separates them: every interval or run range overlaps, too few runs were recorded, no interval was recorded, or more is not better for that metric. The rows are 1 tie and 2 unclear; each row says why.",
      "rows": [
        {
          "metric": "Resolved rate on the same 33 SWE-bench Verified instances",
          "aValue": 0.7576,
          "bValue": 0.697,
          "unit": "rate",
          "aDisplay": "76% (25/33)",
          "bDisplay": "70% (23/33)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Sonnet 4.5 59% to 87%; Claude Opus 4.6 53% to 83%), so this sample cannot separate them.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-same-instance-leaderboard",
          "aN": 33,
          "bN": 33,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "public mini-SWE-agent v2 run, same instances",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.5898,
            0.8717
          ],
          "bRange": [
            0.5266,
            0.8262
          ]
        },
        {
          "metric": "Model calls per instance",
          "aValue": 51,
          "bValue": 28.9,
          "unit": "calls",
          "aDisplay": "51",
          "bDisplay": "28.9",
          "winner": "unclear",
          "basis": "More or fewer calls is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-model-calls",
          "aN": 33,
          "bN": 33,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "public mini-SWE-agent v2 run, same instances"
        },
        {
          "metric": "Recorded cost per resolved instance: Agent vs the public panel",
          "aValue": 0.913,
          "bValue": 0.875,
          "unit": "usd",
          "aDisplay": "$0.91",
          "bDisplay": "$0.88",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.91 vs $0.88) is not tested against run-to-run variation.",
          "studySlug": "cost-thought-experiments",
          "chartId": "cost-per-resolved-agent-vs-panel",
          "aN": 25,
          "bN": 23,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "public mini-SWE-agent v2 run, same instances"
        }
      ]
    },
    {
      "slug": "claude-sonnet-4-5-vs-deepseek-v3-2",
      "a": "claude-sonnet-4-5",
      "b": "deepseek-v3-2",
      "title": "Claude Sonnet 4.5 vs DeepSeek V3.2",
      "seoTitle": "Claude Sonnet 4.5 vs DeepSeek V3.2: measured benchmarks",
      "description": "Claude Sonnet 4.5 vs DeepSeek V3.2: 3 measured metrics from 2 studies, with sample sizes, intervals and every failure counted.",
      "verdict": "Claude Sonnet 4.5 and DeepSeek V3.2 share 3 measured metrics from 2 studies. No row separates them: every interval or run range overlaps, too few runs were recorded, no interval was recorded, or more is not better for that metric. The rows are 1 tie and 2 unclear; each row says why.",
      "rows": [
        {
          "metric": "Resolved rate on the same 33 SWE-bench Verified instances",
          "aValue": 0.7576,
          "bValue": 0.7273,
          "unit": "rate",
          "aDisplay": "76% (25/33)",
          "bDisplay": "73% (24/33)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Sonnet 4.5 59% to 87%; DeepSeek V3.2 56% to 85%), so this sample cannot separate them.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-same-instance-leaderboard",
          "aN": 33,
          "bN": 33,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.5898,
            0.8717
          ],
          "bRange": [
            0.5578,
            0.8493
          ]
        },
        {
          "metric": "Model calls per instance",
          "aValue": 51,
          "bValue": 88.2,
          "unit": "calls",
          "aDisplay": "51",
          "bDisplay": "88.2",
          "winner": "unclear",
          "basis": "More or fewer calls is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-model-calls",
          "aN": 33,
          "bN": 33,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances"
        },
        {
          "metric": "Recorded cost per resolved instance: Agent vs the public panel",
          "aValue": 0.913,
          "bValue": 0.637,
          "unit": "usd",
          "aDisplay": "$0.91",
          "bDisplay": "$0.64",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.91 vs $0.64) is not tested against run-to-run variation.",
          "studySlug": "cost-thought-experiments",
          "chartId": "cost-per-resolved-agent-vs-panel",
          "aN": 25,
          "bN": 24,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances"
        }
      ]
    },
    {
      "slug": "claude-sonnet-4-5-vs-gpt-5-mini",
      "a": "claude-sonnet-4-5",
      "b": "gpt-5-mini",
      "title": "Claude Sonnet 4.5 vs GPT 5 mini",
      "seoTitle": "Claude Sonnet 4.5 vs GPT 5 mini: measured benchmarks",
      "description": "Claude Sonnet 4.5 vs GPT 5 mini: 3 measured metrics from 2 studies, with sample sizes, intervals and every failure counted.",
      "verdict": "Claude Sonnet 4.5 and GPT 5 mini share 3 measured metrics from 2 studies. No row separates them: every interval or run range overlaps, too few runs were recorded, no interval was recorded, or more is not better for that metric. The rows are 1 tie and 2 unclear; each row says why.",
      "rows": [
        {
          "metric": "Resolved rate on the same 33 SWE-bench Verified instances",
          "aValue": 0.7576,
          "bValue": 0.6364,
          "unit": "rate",
          "aDisplay": "76% (25/33)",
          "bDisplay": "64% (21/33)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Sonnet 4.5 59% to 87%; GPT 5 mini 47% to 78%), so this sample cannot separate them.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-same-instance-leaderboard",
          "aN": 33,
          "bN": 33,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "public mini-SWE-agent v2 run, same instances",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.5898,
            0.8717
          ],
          "bRange": [
            0.4662,
            0.7781
          ]
        },
        {
          "metric": "Model calls per instance",
          "aValue": 51,
          "bValue": 20.8,
          "unit": "calls",
          "aDisplay": "51",
          "bDisplay": "20.8",
          "winner": "unclear",
          "basis": "More or fewer calls is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-model-calls",
          "aN": 33,
          "bN": 33,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "public mini-SWE-agent v2 run, same instances"
        },
        {
          "metric": "Recorded cost per resolved instance: Agent vs the public panel",
          "aValue": 0.913,
          "bValue": 0.08,
          "unit": "usd",
          "aDisplay": "$0.91",
          "bDisplay": "$0.080",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.91 vs $0.080, 11x) is not tested against run-to-run variation.",
          "studySlug": "cost-thought-experiments",
          "chartId": "cost-per-resolved-agent-vs-panel",
          "aN": 25,
          "bN": 21,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "public mini-SWE-agent v2 run, same instances"
        }
      ]
    },
    {
      "slug": "claude-sonnet-4-5-vs-kimi-k2-5",
      "a": "claude-sonnet-4-5",
      "b": "kimi-k2-5",
      "title": "Claude Sonnet 4.5 vs Kimi K2.5",
      "seoTitle": "Claude Sonnet 4.5 vs Kimi K2.5: measured benchmarks",
      "description": "Claude Sonnet 4.5 vs Kimi K2.5: 3 measured metrics from 2 studies, with sample sizes, intervals and every failure counted.",
      "verdict": "Claude Sonnet 4.5 and Kimi K2.5 share 3 measured metrics from 2 studies. No row separates them: every interval or run range overlaps, too few runs were recorded, no interval was recorded, or more is not better for that metric. The rows are 1 tie and 2 unclear; each row says why.",
      "rows": [
        {
          "metric": "Resolved rate on the same 33 SWE-bench Verified instances",
          "aValue": 0.7576,
          "bValue": 0.697,
          "unit": "rate",
          "aDisplay": "76% (25/33)",
          "bDisplay": "70% (23/33)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Sonnet 4.5 59% to 87%; Kimi K2.5 53% to 83%), so this sample cannot separate them.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-same-instance-leaderboard",
          "aN": 33,
          "bN": 33,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.5898,
            0.8717
          ],
          "bRange": [
            0.5266,
            0.8262
          ]
        },
        {
          "metric": "Model calls per instance",
          "aValue": 51,
          "bValue": 56.7,
          "unit": "calls",
          "aDisplay": "51",
          "bDisplay": "56.7",
          "winner": "unclear",
          "basis": "More or fewer calls is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-model-calls",
          "aN": 33,
          "bN": 33,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances"
        },
        {
          "metric": "Recorded cost per resolved instance: Agent vs the public panel",
          "aValue": 0.913,
          "bValue": 0.256,
          "unit": "usd",
          "aDisplay": "$0.91",
          "bDisplay": "$0.26",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.91 vs $0.26, 3.6x) is not tested against run-to-run variation.",
          "studySlug": "cost-thought-experiments",
          "chartId": "cost-per-resolved-agent-vs-panel",
          "aN": 25,
          "bN": 23,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances"
        }
      ]
    },
    {
      "slug": "claude-sonnet-4-5-vs-minimax-m2-5",
      "a": "claude-sonnet-4-5",
      "b": "minimax-m2-5",
      "title": "Claude Sonnet 4.5 vs MiniMax M2.5",
      "seoTitle": "Claude Sonnet 4.5 vs MiniMax M2.5: measured benchmarks",
      "description": "Claude Sonnet 4.5 vs MiniMax M2.5: 3 measured metrics from 2 studies, with sample sizes, intervals and every failure counted.",
      "verdict": "Claude Sonnet 4.5 and MiniMax M2.5 share 3 measured metrics from 2 studies. No row separates them: every interval or run range overlaps, too few runs were recorded, no interval was recorded, or more is not better for that metric. The rows are 1 tie and 2 unclear; each row says why.",
      "rows": [
        {
          "metric": "Resolved rate on the same 33 SWE-bench Verified instances",
          "aValue": 0.7576,
          "bValue": 0.697,
          "unit": "rate",
          "aDisplay": "76% (25/33)",
          "bDisplay": "70% (23/33)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Sonnet 4.5 59% to 87%; MiniMax M2.5 53% to 83%), so this sample cannot separate them.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-same-instance-leaderboard",
          "aN": 33,
          "bN": 33,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.5898,
            0.8717
          ],
          "bRange": [
            0.5266,
            0.8262
          ]
        },
        {
          "metric": "Model calls per instance",
          "aValue": 51,
          "bValue": 58.4,
          "unit": "calls",
          "aDisplay": "51",
          "bDisplay": "58.4",
          "winner": "unclear",
          "basis": "More or fewer calls is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-model-calls",
          "aN": 33,
          "bN": 33,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances"
        },
        {
          "metric": "Recorded cost per resolved instance: Agent vs the public panel",
          "aValue": 0.913,
          "bValue": 0.107,
          "unit": "usd",
          "aDisplay": "$0.91",
          "bDisplay": "$0.11",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.91 vs $0.11, 8.5x) is not tested against run-to-run variation.",
          "studySlug": "cost-thought-experiments",
          "chartId": "cost-per-resolved-agent-vs-panel",
          "aN": 25,
          "bN": 23,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances"
        }
      ]
    },
    {
      "slug": "claude-sonnet-5-5-vs-gpt-6-1-sol-openai-api",
      "a": "claude-sonnet-5-5",
      "b": "gpt-6-1-sol-openai-api",
      "title": "Claude Sonnet 5.5 vs GPT-6.1 Sol (OpenAI API)",
      "seoTitle": "Claude Sonnet 5.5 vs GPT-6.1 Sol (OpenAI API): benchmarks",
      "description": "Claude Sonnet 5.5 vs GPT-6.1 Sol (OpenAI API): 3 measured metrics from one study, with sample sizes, intervals and every failure counted.",
      "verdict": "Claude Sonnet 5.5 and GPT-6.1 Sol (OpenAI API) share 3 measured metrics from one study. No row separates them: every interval or run range overlaps, too few runs were recorded, no interval was recorded, or more is not better for that metric. The rows are 3 unclear; each row says why. Every row ran the two sides through different routes (for example Claude Code vs OpenAI API), so they compare route + model pairs, not models alone; the contexts name the route. Some rows rest on small samples (n = 3 at the smallest).",
      "rows": [
        {
          "metric": "Repairing a scheduler: Claude Code vs Codex vs API (Total time)",
          "aValue": 15,
          "bValue": 17.32,
          "unit": "seconds",
          "aDisplay": "15.0 s",
          "bDisplay": "17.3 s",
          "winner": "unclear",
          "basis": "Only 3 runs per side; the run ranges (fastest to slowest) do not overlap (Claude Sonnet 5.5 13.9 s to 15.9 s; GPT-6.1 Sol (OpenAI API) 16.3 s to 18.6 s), but 3 runs cannot show a reliable difference. A range is not a confidence interval.",
          "studySlug": "cli-model-latency-tokens",
          "n": 3,
          "chartId": "scheduler-repair-claude-vs-codex",
          "aN": 3,
          "bN": 3,
          "aContext": "Claude Code CLI · effort medium · scheduler repair, 296 checks, 3 runs",
          "bContext": "OpenAI API · effort medium · scheduler repair, 296 checks, 3 runs",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            13.89,
            15.89
          ],
          "bRange": [
            16.28,
            18.61
          ]
        },
        {
          "metric": "Repairing a scheduler: Claude Code vs Codex vs API (First useful output)",
          "aValue": 7.55,
          "bValue": 7.46,
          "unit": "seconds",
          "aDisplay": "7.55 s",
          "bDisplay": "7.46 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Sonnet 5.5 6.77 s to 7.63 s; GPT-6.1 Sol (OpenAI API) 6.68 s to 9.05 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "cli-model-latency-tokens",
          "n": 3,
          "chartId": "scheduler-repair-claude-vs-codex",
          "aN": 3,
          "bN": 3,
          "aContext": "Claude Code CLI · effort medium · scheduler repair, 296 checks, 3 runs",
          "bContext": "OpenAI API · effort medium · scheduler repair, 296 checks, 3 runs",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            6.77,
            7.63
          ],
          "bRange": [
            6.68,
            9.05
          ]
        },
        {
          "metric": "Output tokens to repair the scheduler (Output tokens)",
          "aValue": 2227,
          "bValue": 1313,
          "unit": "tokens",
          "aDisplay": "2,227",
          "bDisplay": "1,313",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "cli-model-latency-tokens",
          "n": 3,
          "chartId": "scheduler-repair-output-tokens",
          "aN": 3,
          "bN": 3,
          "aContext": "Claude Code CLI · effort medium · scheduler repair, 296 checks, 3 runs",
          "bContext": "OpenAI API · effort medium · scheduler repair, 296 checks, 3 runs"
        }
      ]
    },
    {
      "slug": "deepseek-v3-2-vs-gpt-5-mini",
      "a": "deepseek-v3-2",
      "b": "gpt-5-mini",
      "title": "DeepSeek V3.2 vs GPT 5 mini",
      "seoTitle": "DeepSeek V3.2 vs GPT 5 mini: measured benchmarks",
      "description": "DeepSeek V3.2 vs GPT 5 mini: 3 measured metrics from 2 studies, with sample sizes, intervals and every failure counted.",
      "verdict": "DeepSeek V3.2 and GPT 5 mini share 3 measured metrics from 2 studies. No row separates them: every interval or run range overlaps, too few runs were recorded, no interval was recorded, or more is not better for that metric. The rows are 1 tie and 2 unclear; each row says why.",
      "rows": [
        {
          "metric": "Resolved rate on the same 33 SWE-bench Verified instances",
          "aValue": 0.7273,
          "bValue": 0.6364,
          "unit": "rate",
          "aDisplay": "73% (24/33)",
          "bDisplay": "64% (21/33)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (DeepSeek V3.2 56% to 85%; GPT 5 mini 47% to 78%), so this sample cannot separate them.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-same-instance-leaderboard",
          "aN": 33,
          "bN": 33,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "public mini-SWE-agent v2 run, same instances",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.5578,
            0.8493
          ],
          "bRange": [
            0.4662,
            0.7781
          ]
        },
        {
          "metric": "Model calls per instance",
          "aValue": 88.2,
          "bValue": 20.8,
          "unit": "calls",
          "aDisplay": "88.2",
          "bDisplay": "20.8",
          "winner": "unclear",
          "basis": "More or fewer calls is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-model-calls",
          "aN": 33,
          "bN": 33,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "public mini-SWE-agent v2 run, same instances"
        },
        {
          "metric": "Recorded cost per resolved instance: Agent vs the public panel",
          "aValue": 0.637,
          "bValue": 0.08,
          "unit": "usd",
          "aDisplay": "$0.64",
          "bDisplay": "$0.080",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.64 vs $0.080, 8.0x) is not tested against run-to-run variation.",
          "studySlug": "cost-thought-experiments",
          "chartId": "cost-per-resolved-agent-vs-panel",
          "aN": 24,
          "bN": 21,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "public mini-SWE-agent v2 run, same instances"
        }
      ]
    },
    {
      "slug": "deepseek-v3-2-vs-kimi-k2-5",
      "a": "deepseek-v3-2",
      "b": "kimi-k2-5",
      "title": "DeepSeek V3.2 vs Kimi K2.5",
      "seoTitle": "DeepSeek V3.2 vs Kimi K2.5: measured benchmarks",
      "description": "DeepSeek V3.2 vs Kimi K2.5: 3 measured metrics from 2 studies, with sample sizes, intervals and every failure counted.",
      "verdict": "DeepSeek V3.2 and Kimi K2.5 share 3 measured metrics from 2 studies. No row separates them: every interval or run range overlaps, too few runs were recorded, no interval was recorded, or more is not better for that metric. The rows are 1 tie and 2 unclear; each row says why.",
      "rows": [
        {
          "metric": "Resolved rate on the same 33 SWE-bench Verified instances",
          "aValue": 0.7273,
          "bValue": 0.697,
          "unit": "rate",
          "aDisplay": "73% (24/33)",
          "bDisplay": "70% (23/33)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (DeepSeek V3.2 56% to 85%; Kimi K2.5 53% to 83%), so this sample cannot separate them.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-same-instance-leaderboard",
          "aN": 33,
          "bN": 33,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.5578,
            0.8493
          ],
          "bRange": [
            0.5266,
            0.8262
          ]
        },
        {
          "metric": "Model calls per instance",
          "aValue": 88.2,
          "bValue": 56.7,
          "unit": "calls",
          "aDisplay": "88.2",
          "bDisplay": "56.7",
          "winner": "unclear",
          "basis": "More or fewer calls is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-model-calls",
          "aN": 33,
          "bN": 33,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances"
        },
        {
          "metric": "Recorded cost per resolved instance: Agent vs the public panel",
          "aValue": 0.637,
          "bValue": 0.256,
          "unit": "usd",
          "aDisplay": "$0.64",
          "bDisplay": "$0.26",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.64 vs $0.26, 2.5x) is not tested against run-to-run variation.",
          "studySlug": "cost-thought-experiments",
          "chartId": "cost-per-resolved-agent-vs-panel",
          "aN": 24,
          "bN": 23,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances"
        }
      ]
    },
    {
      "slug": "deepseek-v3-2-vs-minimax-m2-5",
      "a": "deepseek-v3-2",
      "b": "minimax-m2-5",
      "title": "DeepSeek V3.2 vs MiniMax M2.5",
      "seoTitle": "DeepSeek V3.2 vs MiniMax M2.5: measured benchmarks",
      "description": "DeepSeek V3.2 vs MiniMax M2.5: 3 measured metrics from 2 studies, with sample sizes, intervals and every failure counted.",
      "verdict": "DeepSeek V3.2 and MiniMax M2.5 share 3 measured metrics from 2 studies. No row separates them: every interval or run range overlaps, too few runs were recorded, no interval was recorded, or more is not better for that metric. The rows are 1 tie and 2 unclear; each row says why.",
      "rows": [
        {
          "metric": "Resolved rate on the same 33 SWE-bench Verified instances",
          "aValue": 0.7273,
          "bValue": 0.697,
          "unit": "rate",
          "aDisplay": "73% (24/33)",
          "bDisplay": "70% (23/33)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (DeepSeek V3.2 56% to 85%; MiniMax M2.5 53% to 83%), so this sample cannot separate them.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-same-instance-leaderboard",
          "aN": 33,
          "bN": 33,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.5578,
            0.8493
          ],
          "bRange": [
            0.5266,
            0.8262
          ]
        },
        {
          "metric": "Model calls per instance",
          "aValue": 88.2,
          "bValue": 58.4,
          "unit": "calls",
          "aDisplay": "88.2",
          "bDisplay": "58.4",
          "winner": "unclear",
          "basis": "More or fewer calls is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-model-calls",
          "aN": 33,
          "bN": 33,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances"
        },
        {
          "metric": "Recorded cost per resolved instance: Agent vs the public panel",
          "aValue": 0.637,
          "bValue": 0.107,
          "unit": "usd",
          "aDisplay": "$0.64",
          "bDisplay": "$0.11",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.64 vs $0.11, 6.0x) is not tested against run-to-run variation.",
          "studySlug": "cost-thought-experiments",
          "chartId": "cost-per-resolved-agent-vs-panel",
          "aN": 24,
          "bN": 23,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances"
        }
      ]
    },
    {
      "slug": "fireworks-vs-cloudflare",
      "a": "fireworks",
      "b": "cloudflare",
      "title": "Fireworks AI vs Cloudflare Workers AI",
      "seoTitle": "Fireworks AI vs Cloudflare Workers AI: price per model",
      "description": "Fireworks AI vs Cloudflare Workers AI: 3 list prices per million tokens for the same models, as reported by OpenRouter’s public API, with the gap stated.",
      "verdict": "Fireworks AI and Cloudflare Workers AI share 3 reported list prices from one study. No row names a winner: a reported list price has no interval, so each gap is stated, not ranked. Prices change often; the snapshot date is in each row’s context. The rows are 3 ties; each row says why.",
      "rows": [
        {
          "metric": "GLM 5.3: price per million tokens by provider (Input)",
          "aValue": 1.4,
          "bValue": 1.4,
          "unit": "usd",
          "aDisplay": "$1.40",
          "bDisplay": "$1.40",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "aContext": "GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "GLM 5.3: price per million tokens by provider (Output)",
          "aValue": 4.4,
          "bValue": 4.4,
          "unit": "usd",
          "aDisplay": "$4.40",
          "bDisplay": "$4.40",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "aContext": "GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "GLM 5.3: price per million tokens by provider (Cache read)",
          "aValue": 0.26,
          "bValue": 0.26,
          "unit": "usd",
          "aDisplay": "$0.26",
          "bDisplay": "$0.26",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "aContext": "GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "fireworks-vs-novita",
      "a": "fireworks",
      "b": "novita",
      "title": "Fireworks AI vs Novita AI",
      "seoTitle": "Fireworks AI vs Novita AI: price per model",
      "description": "Fireworks AI vs Novita AI: 3 list prices per million tokens for the same models, as reported by OpenRouter’s public API, with the gap stated.",
      "verdict": "Fireworks AI and Novita AI share 3 reported list prices from one study. No row names a winner: a reported list price has no interval, so each gap is stated, not ranked. Prices change often; the snapshot date is in each row’s context. The rows are 3 unclear; each row says why.",
      "rows": [
        {
          "metric": "GLM 5.3: price per million tokens by provider (Input)",
          "aValue": 1.4,
          "bValue": 0.42,
          "unit": "usd",
          "aDisplay": "$1.40",
          "bDisplay": "$0.42",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($1.40 vs $0.42, 3.3x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "aContext": "GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "GLM 5.3: price per million tokens by provider (Output)",
          "aValue": 4.4,
          "bValue": 1.32,
          "unit": "usd",
          "aDisplay": "$4.40",
          "bDisplay": "$1.32",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($4.40 vs $1.32, 3.3x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "aContext": "GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "GLM 5.3: price per million tokens by provider (Cache read)",
          "aValue": 0.26,
          "bValue": 0.078,
          "unit": "usd",
          "aDisplay": "$0.26",
          "bDisplay": "$0.078",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.26 vs $0.078, 3.3x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "aContext": "GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "fireworks-vs-siliconflow",
      "a": "fireworks",
      "b": "siliconflow",
      "title": "Fireworks AI vs SiliconFlow",
      "seoTitle": "Fireworks AI vs SiliconFlow: price per model",
      "description": "Fireworks AI vs SiliconFlow: 3 list prices per million tokens for the same models, as reported by OpenRouter’s public API, with the gap stated.",
      "verdict": "Fireworks AI and SiliconFlow share 3 reported list prices from one study. No row names a winner: a reported list price has no interval, so each gap is stated, not ranked. Prices change often; the snapshot date is in each row’s context. The rows are 3 unclear; each row says why.",
      "rows": [
        {
          "metric": "GLM 5.3: price per million tokens by provider (Input)",
          "aValue": 1.4,
          "bValue": 0.7,
          "unit": "usd",
          "aDisplay": "$1.40",
          "bDisplay": "$0.70",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($1.40 vs $0.70, 2.0x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "aContext": "GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "GLM 5.3: price per million tokens by provider (Output)",
          "aValue": 4.4,
          "bValue": 2.2,
          "unit": "usd",
          "aDisplay": "$4.40",
          "bDisplay": "$2.20",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($4.40 vs $2.20, 2.0x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "aContext": "GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "GLM 5.3: price per million tokens by provider (Cache read)",
          "aValue": 0.26,
          "bValue": 0.13,
          "unit": "usd",
          "aDisplay": "$0.26",
          "bDisplay": "$0.13",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.26 vs $0.13, 2.0x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "aContext": "GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "gemini-3-flash-vs-claude-opus-4-5",
      "a": "gemini-3-flash",
      "b": "claude-opus-4-5",
      "title": "Gemini 3 Flash vs Claude Opus 4.5",
      "seoTitle": "Gemini 3 Flash vs Claude Opus 4.5: measured benchmarks",
      "description": "Gemini 3 Flash vs Claude Opus 4.5: 3 measured metrics from 2 studies, with sample sizes, intervals and every failure counted.",
      "verdict": "Gemini 3 Flash and Claude Opus 4.5 share 3 measured metrics from 2 studies. No row separates them: every interval or run range overlaps, too few runs were recorded, no interval was recorded, or more is not better for that metric. The rows are 1 tie and 2 unclear; each row says why.",
      "rows": [
        {
          "metric": "Resolved rate on the same 33 SWE-bench Verified instances",
          "aValue": 0.8182,
          "bValue": 0.7273,
          "unit": "rate",
          "aDisplay": "82% (27/33)",
          "bDisplay": "73% (24/33)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Gemini 3 Flash 66% to 91%; Claude Opus 4.5 56% to 85%), so this sample cannot separate them.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-same-instance-leaderboard",
          "aN": 33,
          "bN": 33,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.6561,
            0.9139
          ],
          "bRange": [
            0.5578,
            0.8493
          ]
        },
        {
          "metric": "Model calls per instance",
          "aValue": 54.2,
          "bValue": 35.9,
          "unit": "calls",
          "aDisplay": "54.2",
          "bDisplay": "35.9",
          "winner": "unclear",
          "basis": "More or fewer calls is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-model-calls",
          "aN": 33,
          "bN": 33,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances"
        },
        {
          "metric": "Recorded cost per resolved instance: Agent vs the public panel",
          "aValue": 0.436,
          "bValue": 1.184,
          "unit": "usd",
          "aDisplay": "$0.44",
          "bDisplay": "$1.18",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.44 vs $1.18, 2.7x) is not tested against run-to-run variation.",
          "studySlug": "cost-thought-experiments",
          "chartId": "cost-per-resolved-agent-vs-panel",
          "aN": 27,
          "bN": 24,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances"
        }
      ]
    },
    {
      "slug": "gemini-3-flash-vs-claude-opus-4-6",
      "a": "gemini-3-flash",
      "b": "claude-opus-4-6",
      "title": "Gemini 3 Flash vs Claude Opus 4.6",
      "seoTitle": "Gemini 3 Flash vs Claude Opus 4.6: measured benchmarks",
      "description": "Gemini 3 Flash vs Claude Opus 4.6: 3 measured metrics from 2 studies, with sample sizes, intervals and every failure counted.",
      "verdict": "Gemini 3 Flash and Claude Opus 4.6 share 3 measured metrics from 2 studies. No row separates them: every interval or run range overlaps, too few runs were recorded, no interval was recorded, or more is not better for that metric. The rows are 1 tie and 2 unclear; each row says why.",
      "rows": [
        {
          "metric": "Resolved rate on the same 33 SWE-bench Verified instances",
          "aValue": 0.8182,
          "bValue": 0.697,
          "unit": "rate",
          "aDisplay": "82% (27/33)",
          "bDisplay": "70% (23/33)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Gemini 3 Flash 66% to 91%; Claude Opus 4.6 53% to 83%), so this sample cannot separate them.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-same-instance-leaderboard",
          "aN": 33,
          "bN": 33,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "public mini-SWE-agent v2 run, same instances",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.6561,
            0.9139
          ],
          "bRange": [
            0.5266,
            0.8262
          ]
        },
        {
          "metric": "Model calls per instance",
          "aValue": 54.2,
          "bValue": 28.9,
          "unit": "calls",
          "aDisplay": "54.2",
          "bDisplay": "28.9",
          "winner": "unclear",
          "basis": "More or fewer calls is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-model-calls",
          "aN": 33,
          "bN": 33,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "public mini-SWE-agent v2 run, same instances"
        },
        {
          "metric": "Recorded cost per resolved instance: Agent vs the public panel",
          "aValue": 0.436,
          "bValue": 0.875,
          "unit": "usd",
          "aDisplay": "$0.44",
          "bDisplay": "$0.88",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.44 vs $0.88, 2.0x) is not tested against run-to-run variation.",
          "studySlug": "cost-thought-experiments",
          "chartId": "cost-per-resolved-agent-vs-panel",
          "aN": 27,
          "bN": 23,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "public mini-SWE-agent v2 run, same instances"
        }
      ]
    },
    {
      "slug": "gemini-3-flash-vs-claude-sonnet-4-5",
      "a": "gemini-3-flash",
      "b": "claude-sonnet-4-5",
      "title": "Gemini 3 Flash vs Claude Sonnet 4.5",
      "seoTitle": "Gemini 3 Flash vs Claude Sonnet 4.5: measured benchmarks",
      "description": "Gemini 3 Flash vs Claude Sonnet 4.5: 3 measured metrics from 2 studies, with sample sizes, intervals and every failure counted.",
      "verdict": "Gemini 3 Flash and Claude Sonnet 4.5 share 3 measured metrics from 2 studies. No row separates them: every interval or run range overlaps, too few runs were recorded, no interval was recorded, or more is not better for that metric. The rows are 1 tie and 2 unclear; each row says why.",
      "rows": [
        {
          "metric": "Resolved rate on the same 33 SWE-bench Verified instances",
          "aValue": 0.8182,
          "bValue": 0.7576,
          "unit": "rate",
          "aDisplay": "82% (27/33)",
          "bDisplay": "76% (25/33)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Gemini 3 Flash 66% to 91%; Claude Sonnet 4.5 59% to 87%), so this sample cannot separate them.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-same-instance-leaderboard",
          "aN": 33,
          "bN": 33,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.6561,
            0.9139
          ],
          "bRange": [
            0.5898,
            0.8717
          ]
        },
        {
          "metric": "Model calls per instance",
          "aValue": 54.2,
          "bValue": 51,
          "unit": "calls",
          "aDisplay": "54.2",
          "bDisplay": "51",
          "winner": "unclear",
          "basis": "More or fewer calls is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-model-calls",
          "aN": 33,
          "bN": 33,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances"
        },
        {
          "metric": "Recorded cost per resolved instance: Agent vs the public panel",
          "aValue": 0.436,
          "bValue": 0.913,
          "unit": "usd",
          "aDisplay": "$0.44",
          "bDisplay": "$0.91",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.44 vs $0.91, 2.1x) is not tested against run-to-run variation.",
          "studySlug": "cost-thought-experiments",
          "chartId": "cost-per-resolved-agent-vs-panel",
          "aN": 27,
          "bN": 25,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances"
        }
      ]
    },
    {
      "slug": "gemini-3-flash-vs-deepseek-v3-2",
      "a": "gemini-3-flash",
      "b": "deepseek-v3-2",
      "title": "Gemini 3 Flash vs DeepSeek V3.2",
      "seoTitle": "Gemini 3 Flash vs DeepSeek V3.2: measured benchmarks",
      "description": "Gemini 3 Flash vs DeepSeek V3.2: 3 measured metrics from 2 studies, with sample sizes, intervals and every failure counted.",
      "verdict": "Gemini 3 Flash and DeepSeek V3.2 share 3 measured metrics from 2 studies. No row separates them: every interval or run range overlaps, too few runs were recorded, no interval was recorded, or more is not better for that metric. The rows are 1 tie and 2 unclear; each row says why.",
      "rows": [
        {
          "metric": "Resolved rate on the same 33 SWE-bench Verified instances",
          "aValue": 0.8182,
          "bValue": 0.7273,
          "unit": "rate",
          "aDisplay": "82% (27/33)",
          "bDisplay": "73% (24/33)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Gemini 3 Flash 66% to 91%; DeepSeek V3.2 56% to 85%), so this sample cannot separate them.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-same-instance-leaderboard",
          "aN": 33,
          "bN": 33,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.6561,
            0.9139
          ],
          "bRange": [
            0.5578,
            0.8493
          ]
        },
        {
          "metric": "Model calls per instance",
          "aValue": 54.2,
          "bValue": 88.2,
          "unit": "calls",
          "aDisplay": "54.2",
          "bDisplay": "88.2",
          "winner": "unclear",
          "basis": "More or fewer calls is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-model-calls",
          "aN": 33,
          "bN": 33,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances"
        },
        {
          "metric": "Recorded cost per resolved instance: Agent vs the public panel",
          "aValue": 0.436,
          "bValue": 0.637,
          "unit": "usd",
          "aDisplay": "$0.44",
          "bDisplay": "$0.64",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.44 vs $0.64) is not tested against run-to-run variation.",
          "studySlug": "cost-thought-experiments",
          "chartId": "cost-per-resolved-agent-vs-panel",
          "aN": 27,
          "bN": 24,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances"
        }
      ]
    },
    {
      "slug": "gemini-3-flash-vs-glm-5",
      "a": "gemini-3-flash",
      "b": "glm-5",
      "title": "Gemini 3 Flash vs GLM 5",
      "seoTitle": "Gemini 3 Flash vs GLM 5: measured benchmarks",
      "description": "Gemini 3 Flash vs GLM 5: 3 measured metrics from 2 studies (Resolved rate on the same 33 SWE-bench Verified instances; more), with sample sizes and intervals.",
      "verdict": "Gemini 3 Flash and GLM 5 share 3 measured metrics from 2 studies. No row separates them: every interval or run range overlaps, too few runs were recorded, no interval was recorded, or more is not better for that metric. The rows are 1 tie and 2 unclear; each row says why.",
      "rows": [
        {
          "metric": "Resolved rate on the same 33 SWE-bench Verified instances",
          "aValue": 0.8182,
          "bValue": 0.7879,
          "unit": "rate",
          "aDisplay": "82% (27/33)",
          "bDisplay": "79% (26/33)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Gemini 3 Flash 66% to 91%; GLM 5 62% to 89%), so this sample cannot separate them.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-same-instance-leaderboard",
          "aN": 33,
          "bN": 33,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.6561,
            0.9139
          ],
          "bRange": [
            0.6225,
            0.8932
          ]
        },
        {
          "metric": "Model calls per instance",
          "aValue": 54.2,
          "bValue": 77.5,
          "unit": "calls",
          "aDisplay": "54.2",
          "bDisplay": "77.5",
          "winner": "unclear",
          "basis": "More or fewer calls is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-model-calls",
          "aN": 33,
          "bN": 33,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances"
        },
        {
          "metric": "Recorded cost per resolved instance: Agent vs the public panel",
          "aValue": 0.436,
          "bValue": 0.667,
          "unit": "usd",
          "aDisplay": "$0.44",
          "bDisplay": "$0.67",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.44 vs $0.67) is not tested against run-to-run variation.",
          "studySlug": "cost-thought-experiments",
          "chartId": "cost-per-resolved-agent-vs-panel",
          "aN": 27,
          "bN": 26,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances"
        }
      ]
    },
    {
      "slug": "gemini-3-flash-vs-gpt-5-mini",
      "a": "gemini-3-flash",
      "b": "gpt-5-mini",
      "title": "Gemini 3 Flash vs GPT 5 mini",
      "seoTitle": "Gemini 3 Flash vs GPT 5 mini: measured benchmarks",
      "description": "Gemini 3 Flash vs GPT 5 mini: 3 measured metrics from 2 studies, with sample sizes, intervals and every failure counted.",
      "verdict": "Gemini 3 Flash and GPT 5 mini share 3 measured metrics from 2 studies. No row separates them: every interval or run range overlaps, too few runs were recorded, no interval was recorded, or more is not better for that metric. The rows are 1 tie and 2 unclear; each row says why.",
      "rows": [
        {
          "metric": "Resolved rate on the same 33 SWE-bench Verified instances",
          "aValue": 0.8182,
          "bValue": 0.6364,
          "unit": "rate",
          "aDisplay": "82% (27/33)",
          "bDisplay": "64% (21/33)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Gemini 3 Flash 66% to 91%; GPT 5 mini 47% to 78%), so this sample cannot separate them.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-same-instance-leaderboard",
          "aN": 33,
          "bN": 33,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "public mini-SWE-agent v2 run, same instances",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.6561,
            0.9139
          ],
          "bRange": [
            0.4662,
            0.7781
          ]
        },
        {
          "metric": "Model calls per instance",
          "aValue": 54.2,
          "bValue": 20.8,
          "unit": "calls",
          "aDisplay": "54.2",
          "bDisplay": "20.8",
          "winner": "unclear",
          "basis": "More or fewer calls is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-model-calls",
          "aN": 33,
          "bN": 33,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "public mini-SWE-agent v2 run, same instances"
        },
        {
          "metric": "Recorded cost per resolved instance: Agent vs the public panel",
          "aValue": 0.436,
          "bValue": 0.08,
          "unit": "usd",
          "aDisplay": "$0.44",
          "bDisplay": "$0.080",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.44 vs $0.080, 5.5x) is not tested against run-to-run variation.",
          "studySlug": "cost-thought-experiments",
          "chartId": "cost-per-resolved-agent-vs-panel",
          "aN": 27,
          "bN": 21,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "public mini-SWE-agent v2 run, same instances"
        }
      ]
    },
    {
      "slug": "gemini-3-flash-vs-kimi-k2-5",
      "a": "gemini-3-flash",
      "b": "kimi-k2-5",
      "title": "Gemini 3 Flash vs Kimi K2.5",
      "seoTitle": "Gemini 3 Flash vs Kimi K2.5: measured benchmarks",
      "description": "Gemini 3 Flash vs Kimi K2.5: 3 measured metrics from 2 studies, with sample sizes, intervals and every failure counted.",
      "verdict": "Gemini 3 Flash and Kimi K2.5 share 3 measured metrics from 2 studies. No row separates them: every interval or run range overlaps, too few runs were recorded, no interval was recorded, or more is not better for that metric. The rows are 1 tie and 2 unclear; each row says why.",
      "rows": [
        {
          "metric": "Resolved rate on the same 33 SWE-bench Verified instances",
          "aValue": 0.8182,
          "bValue": 0.697,
          "unit": "rate",
          "aDisplay": "82% (27/33)",
          "bDisplay": "70% (23/33)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Gemini 3 Flash 66% to 91%; Kimi K2.5 53% to 83%), so this sample cannot separate them.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-same-instance-leaderboard",
          "aN": 33,
          "bN": 33,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.6561,
            0.9139
          ],
          "bRange": [
            0.5266,
            0.8262
          ]
        },
        {
          "metric": "Model calls per instance",
          "aValue": 54.2,
          "bValue": 56.7,
          "unit": "calls",
          "aDisplay": "54.2",
          "bDisplay": "56.7",
          "winner": "unclear",
          "basis": "More or fewer calls is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-model-calls",
          "aN": 33,
          "bN": 33,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances"
        },
        {
          "metric": "Recorded cost per resolved instance: Agent vs the public panel",
          "aValue": 0.436,
          "bValue": 0.256,
          "unit": "usd",
          "aDisplay": "$0.44",
          "bDisplay": "$0.26",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.44 vs $0.26) is not tested against run-to-run variation.",
          "studySlug": "cost-thought-experiments",
          "chartId": "cost-per-resolved-agent-vs-panel",
          "aN": 27,
          "bN": 23,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances"
        }
      ]
    },
    {
      "slug": "gemini-3-flash-vs-minimax-m2-5",
      "a": "gemini-3-flash",
      "b": "minimax-m2-5",
      "title": "Gemini 3 Flash vs MiniMax M2.5",
      "seoTitle": "Gemini 3 Flash vs MiniMax M2.5: measured benchmarks",
      "description": "Gemini 3 Flash vs MiniMax M2.5: 3 measured metrics from 2 studies, with sample sizes, intervals and every failure counted.",
      "verdict": "Gemini 3 Flash and MiniMax M2.5 share 3 measured metrics from 2 studies. No row separates them: every interval or run range overlaps, too few runs were recorded, no interval was recorded, or more is not better for that metric. The rows are 1 tie and 2 unclear; each row says why.",
      "rows": [
        {
          "metric": "Resolved rate on the same 33 SWE-bench Verified instances",
          "aValue": 0.8182,
          "bValue": 0.697,
          "unit": "rate",
          "aDisplay": "82% (27/33)",
          "bDisplay": "70% (23/33)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Gemini 3 Flash 66% to 91%; MiniMax M2.5 53% to 83%), so this sample cannot separate them.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-same-instance-leaderboard",
          "aN": 33,
          "bN": 33,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.6561,
            0.9139
          ],
          "bRange": [
            0.5266,
            0.8262
          ]
        },
        {
          "metric": "Model calls per instance",
          "aValue": 54.2,
          "bValue": 58.4,
          "unit": "calls",
          "aDisplay": "54.2",
          "bDisplay": "58.4",
          "winner": "unclear",
          "basis": "More or fewer calls is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-model-calls",
          "aN": 33,
          "bN": 33,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances"
        },
        {
          "metric": "Recorded cost per resolved instance: Agent vs the public panel",
          "aValue": 0.436,
          "bValue": 0.107,
          "unit": "usd",
          "aDisplay": "$0.44",
          "bDisplay": "$0.11",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.44 vs $0.11, 4.1x) is not tested against run-to-run variation.",
          "studySlug": "cost-thought-experiments",
          "chartId": "cost-per-resolved-agent-vs-panel",
          "aN": 27,
          "bN": 23,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances"
        }
      ]
    },
    {
      "slug": "glm-5-vs-claude-opus-4-5",
      "a": "glm-5",
      "b": "claude-opus-4-5",
      "title": "GLM 5 vs Claude Opus 4.5",
      "seoTitle": "GLM 5 vs Claude Opus 4.5: measured benchmarks",
      "description": "GLM 5 vs Claude Opus 4.5: 3 measured metrics from 2 studies (Resolved rate on the same 33 SWE-bench Verified instances; more), with sample sizes and intervals.",
      "verdict": "GLM 5 and Claude Opus 4.5 share 3 measured metrics from 2 studies. No row separates them: every interval or run range overlaps, too few runs were recorded, no interval was recorded, or more is not better for that metric. The rows are 1 tie and 2 unclear; each row says why.",
      "rows": [
        {
          "metric": "Resolved rate on the same 33 SWE-bench Verified instances",
          "aValue": 0.7879,
          "bValue": 0.7273,
          "unit": "rate",
          "aDisplay": "79% (26/33)",
          "bDisplay": "73% (24/33)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (GLM 5 62% to 89%; Claude Opus 4.5 56% to 85%), so this sample cannot separate them.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-same-instance-leaderboard",
          "aN": 33,
          "bN": 33,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.6225,
            0.8932
          ],
          "bRange": [
            0.5578,
            0.8493
          ]
        },
        {
          "metric": "Model calls per instance",
          "aValue": 77.5,
          "bValue": 35.9,
          "unit": "calls",
          "aDisplay": "77.5",
          "bDisplay": "35.9",
          "winner": "unclear",
          "basis": "More or fewer calls is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-model-calls",
          "aN": 33,
          "bN": 33,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances"
        },
        {
          "metric": "Recorded cost per resolved instance: Agent vs the public panel",
          "aValue": 0.667,
          "bValue": 1.184,
          "unit": "usd",
          "aDisplay": "$0.67",
          "bDisplay": "$1.18",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.67 vs $1.18) is not tested against run-to-run variation.",
          "studySlug": "cost-thought-experiments",
          "chartId": "cost-per-resolved-agent-vs-panel",
          "aN": 26,
          "bN": 24,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances"
        }
      ]
    },
    {
      "slug": "glm-5-vs-claude-opus-4-6",
      "a": "glm-5",
      "b": "claude-opus-4-6",
      "title": "GLM 5 vs Claude Opus 4.6",
      "seoTitle": "GLM 5 vs Claude Opus 4.6: measured benchmarks",
      "description": "GLM 5 vs Claude Opus 4.6: 3 measured metrics from 2 studies (Resolved rate on the same 33 SWE-bench Verified instances; more), with sample sizes and intervals.",
      "verdict": "GLM 5 and Claude Opus 4.6 share 3 measured metrics from 2 studies. No row separates them: every interval or run range overlaps, too few runs were recorded, no interval was recorded, or more is not better for that metric. The rows are 1 tie and 2 unclear; each row says why.",
      "rows": [
        {
          "metric": "Resolved rate on the same 33 SWE-bench Verified instances",
          "aValue": 0.7879,
          "bValue": 0.697,
          "unit": "rate",
          "aDisplay": "79% (26/33)",
          "bDisplay": "70% (23/33)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (GLM 5 62% to 89%; Claude Opus 4.6 53% to 83%), so this sample cannot separate them.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-same-instance-leaderboard",
          "aN": 33,
          "bN": 33,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "public mini-SWE-agent v2 run, same instances",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.6225,
            0.8932
          ],
          "bRange": [
            0.5266,
            0.8262
          ]
        },
        {
          "metric": "Model calls per instance",
          "aValue": 77.5,
          "bValue": 28.9,
          "unit": "calls",
          "aDisplay": "77.5",
          "bDisplay": "28.9",
          "winner": "unclear",
          "basis": "More or fewer calls is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-model-calls",
          "aN": 33,
          "bN": 33,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "public mini-SWE-agent v2 run, same instances"
        },
        {
          "metric": "Recorded cost per resolved instance: Agent vs the public panel",
          "aValue": 0.667,
          "bValue": 0.875,
          "unit": "usd",
          "aDisplay": "$0.67",
          "bDisplay": "$0.88",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.67 vs $0.88) is not tested against run-to-run variation.",
          "studySlug": "cost-thought-experiments",
          "chartId": "cost-per-resolved-agent-vs-panel",
          "aN": 26,
          "bN": 23,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "public mini-SWE-agent v2 run, same instances"
        }
      ]
    },
    {
      "slug": "glm-5-vs-claude-sonnet-4-5",
      "a": "glm-5",
      "b": "claude-sonnet-4-5",
      "title": "GLM 5 vs Claude Sonnet 4.5",
      "seoTitle": "GLM 5 vs Claude Sonnet 4.5: measured benchmarks",
      "description": "GLM 5 vs Claude Sonnet 4.5: 3 measured metrics from 2 studies, with sample sizes, intervals and every failure counted.",
      "verdict": "GLM 5 and Claude Sonnet 4.5 share 3 measured metrics from 2 studies. No row separates them: every interval or run range overlaps, too few runs were recorded, no interval was recorded, or more is not better for that metric. The rows are 1 tie and 2 unclear; each row says why.",
      "rows": [
        {
          "metric": "Resolved rate on the same 33 SWE-bench Verified instances",
          "aValue": 0.7879,
          "bValue": 0.7576,
          "unit": "rate",
          "aDisplay": "79% (26/33)",
          "bDisplay": "76% (25/33)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (GLM 5 62% to 89%; Claude Sonnet 4.5 59% to 87%), so this sample cannot separate them.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-same-instance-leaderboard",
          "aN": 33,
          "bN": 33,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.6225,
            0.8932
          ],
          "bRange": [
            0.5898,
            0.8717
          ]
        },
        {
          "metric": "Model calls per instance",
          "aValue": 77.5,
          "bValue": 51,
          "unit": "calls",
          "aDisplay": "77.5",
          "bDisplay": "51",
          "winner": "unclear",
          "basis": "More or fewer calls is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-model-calls",
          "aN": 33,
          "bN": 33,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances"
        },
        {
          "metric": "Recorded cost per resolved instance: Agent vs the public panel",
          "aValue": 0.667,
          "bValue": 0.913,
          "unit": "usd",
          "aDisplay": "$0.67",
          "bDisplay": "$0.91",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.67 vs $0.91) is not tested against run-to-run variation.",
          "studySlug": "cost-thought-experiments",
          "chartId": "cost-per-resolved-agent-vs-panel",
          "aN": 26,
          "bN": 25,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances"
        }
      ]
    },
    {
      "slug": "glm-5-vs-deepseek-v3-2",
      "a": "glm-5",
      "b": "deepseek-v3-2",
      "title": "GLM 5 vs DeepSeek V3.2",
      "seoTitle": "GLM 5 vs DeepSeek V3.2: measured benchmarks",
      "description": "GLM 5 vs DeepSeek V3.2: 3 measured metrics from 2 studies (Resolved rate on the same 33 SWE-bench Verified instances; more), with sample sizes and intervals.",
      "verdict": "GLM 5 and DeepSeek V3.2 share 3 measured metrics from 2 studies. No row separates them: every interval or run range overlaps, too few runs were recorded, no interval was recorded, or more is not better for that metric. The rows are 1 tie and 2 unclear; each row says why.",
      "rows": [
        {
          "metric": "Resolved rate on the same 33 SWE-bench Verified instances",
          "aValue": 0.7879,
          "bValue": 0.7273,
          "unit": "rate",
          "aDisplay": "79% (26/33)",
          "bDisplay": "73% (24/33)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (GLM 5 62% to 89%; DeepSeek V3.2 56% to 85%), so this sample cannot separate them.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-same-instance-leaderboard",
          "aN": 33,
          "bN": 33,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.6225,
            0.8932
          ],
          "bRange": [
            0.5578,
            0.8493
          ]
        },
        {
          "metric": "Model calls per instance",
          "aValue": 77.5,
          "bValue": 88.2,
          "unit": "calls",
          "aDisplay": "77.5",
          "bDisplay": "88.2",
          "winner": "unclear",
          "basis": "More or fewer calls is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-model-calls",
          "aN": 33,
          "bN": 33,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances"
        },
        {
          "metric": "Recorded cost per resolved instance: Agent vs the public panel",
          "aValue": 0.667,
          "bValue": 0.637,
          "unit": "usd",
          "aDisplay": "$0.67",
          "bDisplay": "$0.64",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.67 vs $0.64) is not tested against run-to-run variation.",
          "studySlug": "cost-thought-experiments",
          "chartId": "cost-per-resolved-agent-vs-panel",
          "aN": 26,
          "bN": 24,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances"
        }
      ]
    },
    {
      "slug": "glm-5-vs-gpt-5-mini",
      "a": "glm-5",
      "b": "gpt-5-mini",
      "title": "GLM 5 vs GPT 5 mini",
      "seoTitle": "GLM 5 vs GPT 5 mini: measured benchmarks",
      "description": "GLM 5 vs GPT 5 mini: 3 measured metrics from 2 studies (Resolved rate on the same 33 SWE-bench Verified instances; more), with sample sizes and intervals.",
      "verdict": "GLM 5 and GPT 5 mini share 3 measured metrics from 2 studies. No row separates them: every interval or run range overlaps, too few runs were recorded, no interval was recorded, or more is not better for that metric. The rows are 1 tie and 2 unclear; each row says why.",
      "rows": [
        {
          "metric": "Resolved rate on the same 33 SWE-bench Verified instances",
          "aValue": 0.7879,
          "bValue": 0.6364,
          "unit": "rate",
          "aDisplay": "79% (26/33)",
          "bDisplay": "64% (21/33)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (GLM 5 62% to 89%; GPT 5 mini 47% to 78%), so this sample cannot separate them.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-same-instance-leaderboard",
          "aN": 33,
          "bN": 33,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "public mini-SWE-agent v2 run, same instances",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.6225,
            0.8932
          ],
          "bRange": [
            0.4662,
            0.7781
          ]
        },
        {
          "metric": "Model calls per instance",
          "aValue": 77.5,
          "bValue": 20.8,
          "unit": "calls",
          "aDisplay": "77.5",
          "bDisplay": "20.8",
          "winner": "unclear",
          "basis": "More or fewer calls is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-model-calls",
          "aN": 33,
          "bN": 33,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "public mini-SWE-agent v2 run, same instances"
        },
        {
          "metric": "Recorded cost per resolved instance: Agent vs the public panel",
          "aValue": 0.667,
          "bValue": 0.08,
          "unit": "usd",
          "aDisplay": "$0.67",
          "bDisplay": "$0.080",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.67 vs $0.080, 8.3x) is not tested against run-to-run variation.",
          "studySlug": "cost-thought-experiments",
          "chartId": "cost-per-resolved-agent-vs-panel",
          "aN": 26,
          "bN": 21,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "public mini-SWE-agent v2 run, same instances"
        }
      ]
    },
    {
      "slug": "glm-5-vs-kimi-k2-5",
      "a": "glm-5",
      "b": "kimi-k2-5",
      "title": "GLM 5 vs Kimi K2.5",
      "seoTitle": "GLM 5 vs Kimi K2.5: measured benchmarks",
      "description": "GLM 5 vs Kimi K2.5: 3 measured metrics from 2 studies (Resolved rate on the same 33 SWE-bench Verified instances; more), with sample sizes and intervals.",
      "verdict": "GLM 5 and Kimi K2.5 share 3 measured metrics from 2 studies. No row separates them: every interval or run range overlaps, too few runs were recorded, no interval was recorded, or more is not better for that metric. The rows are 1 tie and 2 unclear; each row says why.",
      "rows": [
        {
          "metric": "Resolved rate on the same 33 SWE-bench Verified instances",
          "aValue": 0.7879,
          "bValue": 0.697,
          "unit": "rate",
          "aDisplay": "79% (26/33)",
          "bDisplay": "70% (23/33)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (GLM 5 62% to 89%; Kimi K2.5 53% to 83%), so this sample cannot separate them.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-same-instance-leaderboard",
          "aN": 33,
          "bN": 33,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.6225,
            0.8932
          ],
          "bRange": [
            0.5266,
            0.8262
          ]
        },
        {
          "metric": "Model calls per instance",
          "aValue": 77.5,
          "bValue": 56.7,
          "unit": "calls",
          "aDisplay": "77.5",
          "bDisplay": "56.7",
          "winner": "unclear",
          "basis": "More or fewer calls is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-model-calls",
          "aN": 33,
          "bN": 33,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances"
        },
        {
          "metric": "Recorded cost per resolved instance: Agent vs the public panel",
          "aValue": 0.667,
          "bValue": 0.256,
          "unit": "usd",
          "aDisplay": "$0.67",
          "bDisplay": "$0.26",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.67 vs $0.26, 2.6x) is not tested against run-to-run variation.",
          "studySlug": "cost-thought-experiments",
          "chartId": "cost-per-resolved-agent-vs-panel",
          "aN": 26,
          "bN": 23,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances"
        }
      ]
    },
    {
      "slug": "glm-5-vs-minimax-m2-5",
      "a": "glm-5",
      "b": "minimax-m2-5",
      "title": "GLM 5 vs MiniMax M2.5",
      "seoTitle": "GLM 5 vs MiniMax M2.5: measured benchmarks",
      "description": "GLM 5 vs MiniMax M2.5: 3 measured metrics from 2 studies (Resolved rate on the same 33 SWE-bench Verified instances; more), with sample sizes and intervals.",
      "verdict": "GLM 5 and MiniMax M2.5 share 3 measured metrics from 2 studies. No row separates them: every interval or run range overlaps, too few runs were recorded, no interval was recorded, or more is not better for that metric. The rows are 1 tie and 2 unclear; each row says why.",
      "rows": [
        {
          "metric": "Resolved rate on the same 33 SWE-bench Verified instances",
          "aValue": 0.7879,
          "bValue": 0.697,
          "unit": "rate",
          "aDisplay": "79% (26/33)",
          "bDisplay": "70% (23/33)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (GLM 5 62% to 89%; MiniMax M2.5 53% to 83%), so this sample cannot separate them.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-same-instance-leaderboard",
          "aN": 33,
          "bN": 33,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.6225,
            0.8932
          ],
          "bRange": [
            0.5266,
            0.8262
          ]
        },
        {
          "metric": "Model calls per instance",
          "aValue": 77.5,
          "bValue": 58.4,
          "unit": "calls",
          "aDisplay": "77.5",
          "bDisplay": "58.4",
          "winner": "unclear",
          "basis": "More or fewer calls is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-model-calls",
          "aN": 33,
          "bN": 33,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances"
        },
        {
          "metric": "Recorded cost per resolved instance: Agent vs the public panel",
          "aValue": 0.667,
          "bValue": 0.107,
          "unit": "usd",
          "aDisplay": "$0.67",
          "bDisplay": "$0.11",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.67 vs $0.11, 6.2x) is not tested against run-to-run variation.",
          "studySlug": "cost-thought-experiments",
          "chartId": "cost-per-resolved-agent-vs-panel",
          "aN": 26,
          "bN": 23,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances"
        }
      ]
    },
    {
      "slug": "gpt-5-2-vs-claude-opus-4-5",
      "a": "gpt-5-2",
      "b": "claude-opus-4-5",
      "title": "GPT 5.2 vs Claude Opus 4.5",
      "seoTitle": "GPT 5.2 vs Claude Opus 4.5: measured benchmarks",
      "description": "GPT 5.2 vs Claude Opus 4.5: 3 measured metrics from 2 studies, with sample sizes, intervals and every failure counted.",
      "verdict": "GPT 5.2 and Claude Opus 4.5 share 3 measured metrics from 2 studies. No row separates them: every interval or run range overlaps, too few runs were recorded, no interval was recorded, or more is not better for that metric. The rows are 1 tie and 2 unclear; each row says why.",
      "rows": [
        {
          "metric": "Resolved rate on the same 33 SWE-bench Verified instances",
          "aValue": 0.8485,
          "bValue": 0.7273,
          "unit": "rate",
          "aDisplay": "85% (28/33)",
          "bDisplay": "73% (24/33)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (GPT 5.2 69% to 93%; Claude Opus 4.5 56% to 85%), so this sample cannot separate them.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-same-instance-leaderboard",
          "aN": 33,
          "bN": 33,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.6908,
            0.9335
          ],
          "bRange": [
            0.5578,
            0.8493
          ]
        },
        {
          "metric": "Model calls per instance",
          "aValue": 35.6,
          "bValue": 35.9,
          "unit": "calls",
          "aDisplay": "35.6",
          "bDisplay": "35.9",
          "winner": "unclear",
          "basis": "More or fewer calls is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-model-calls",
          "aN": 33,
          "bN": 33,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances"
        },
        {
          "metric": "Recorded cost per resolved instance: Agent vs the public panel",
          "aValue": 0.628,
          "bValue": 1.184,
          "unit": "usd",
          "aDisplay": "$0.63",
          "bDisplay": "$1.18",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.63 vs $1.18) is not tested against run-to-run variation.",
          "studySlug": "cost-thought-experiments",
          "chartId": "cost-per-resolved-agent-vs-panel",
          "aN": 28,
          "bN": 24,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances"
        }
      ]
    },
    {
      "slug": "gpt-5-2-vs-claude-opus-4-6",
      "a": "gpt-5-2",
      "b": "claude-opus-4-6",
      "title": "GPT 5.2 vs Claude Opus 4.6",
      "seoTitle": "GPT 5.2 vs Claude Opus 4.6: measured benchmarks",
      "description": "GPT 5.2 vs Claude Opus 4.6: 3 measured metrics from 2 studies, with sample sizes, intervals and every failure counted.",
      "verdict": "GPT 5.2 and Claude Opus 4.6 share 3 measured metrics from 2 studies. No row separates them: every interval or run range overlaps, too few runs were recorded, no interval was recorded, or more is not better for that metric. The rows are 1 tie and 2 unclear; each row says why.",
      "rows": [
        {
          "metric": "Resolved rate on the same 33 SWE-bench Verified instances",
          "aValue": 0.8485,
          "bValue": 0.697,
          "unit": "rate",
          "aDisplay": "85% (28/33)",
          "bDisplay": "70% (23/33)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (GPT 5.2 69% to 93%; Claude Opus 4.6 53% to 83%), so this sample cannot separate them.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-same-instance-leaderboard",
          "aN": 33,
          "bN": 33,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "public mini-SWE-agent v2 run, same instances",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.6908,
            0.9335
          ],
          "bRange": [
            0.5266,
            0.8262
          ]
        },
        {
          "metric": "Model calls per instance",
          "aValue": 35.6,
          "bValue": 28.9,
          "unit": "calls",
          "aDisplay": "35.6",
          "bDisplay": "28.9",
          "winner": "unclear",
          "basis": "More or fewer calls is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-model-calls",
          "aN": 33,
          "bN": 33,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "public mini-SWE-agent v2 run, same instances"
        },
        {
          "metric": "Recorded cost per resolved instance: Agent vs the public panel",
          "aValue": 0.628,
          "bValue": 0.875,
          "unit": "usd",
          "aDisplay": "$0.63",
          "bDisplay": "$0.88",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.63 vs $0.88) is not tested against run-to-run variation.",
          "studySlug": "cost-thought-experiments",
          "chartId": "cost-per-resolved-agent-vs-panel",
          "aN": 28,
          "bN": 23,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "public mini-SWE-agent v2 run, same instances"
        }
      ]
    },
    {
      "slug": "gpt-5-2-vs-claude-sonnet-4-5",
      "a": "gpt-5-2",
      "b": "claude-sonnet-4-5",
      "title": "GPT 5.2 vs Claude Sonnet 4.5",
      "seoTitle": "GPT 5.2 vs Claude Sonnet 4.5: measured benchmarks",
      "description": "GPT 5.2 vs Claude Sonnet 4.5: 3 measured metrics from 2 studies, with sample sizes, intervals and every failure counted.",
      "verdict": "GPT 5.2 and Claude Sonnet 4.5 share 3 measured metrics from 2 studies. No row separates them: every interval or run range overlaps, too few runs were recorded, no interval was recorded, or more is not better for that metric. The rows are 1 tie and 2 unclear; each row says why.",
      "rows": [
        {
          "metric": "Resolved rate on the same 33 SWE-bench Verified instances",
          "aValue": 0.8485,
          "bValue": 0.7576,
          "unit": "rate",
          "aDisplay": "85% (28/33)",
          "bDisplay": "76% (25/33)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (GPT 5.2 69% to 93%; Claude Sonnet 4.5 59% to 87%), so this sample cannot separate them.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-same-instance-leaderboard",
          "aN": 33,
          "bN": 33,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.6908,
            0.9335
          ],
          "bRange": [
            0.5898,
            0.8717
          ]
        },
        {
          "metric": "Model calls per instance",
          "aValue": 35.6,
          "bValue": 51,
          "unit": "calls",
          "aDisplay": "35.6",
          "bDisplay": "51",
          "winner": "unclear",
          "basis": "More or fewer calls is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-model-calls",
          "aN": 33,
          "bN": 33,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances"
        },
        {
          "metric": "Recorded cost per resolved instance: Agent vs the public panel",
          "aValue": 0.628,
          "bValue": 0.913,
          "unit": "usd",
          "aDisplay": "$0.63",
          "bDisplay": "$0.91",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.63 vs $0.91) is not tested against run-to-run variation.",
          "studySlug": "cost-thought-experiments",
          "chartId": "cost-per-resolved-agent-vs-panel",
          "aN": 28,
          "bN": 25,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances"
        }
      ]
    },
    {
      "slug": "gpt-5-2-vs-deepseek-v3-2",
      "a": "gpt-5-2",
      "b": "deepseek-v3-2",
      "title": "GPT 5.2 vs DeepSeek V3.2",
      "seoTitle": "GPT 5.2 vs DeepSeek V3.2: measured benchmarks",
      "description": "GPT 5.2 vs DeepSeek V3.2: 3 measured metrics from 2 studies (Resolved rate on the same 33 SWE-bench Verified instances; more), with sample sizes and intervals.",
      "verdict": "GPT 5.2 and DeepSeek V3.2 share 3 measured metrics from 2 studies. No row separates them: every interval or run range overlaps, too few runs were recorded, no interval was recorded, or more is not better for that metric. The rows are 1 tie and 2 unclear; each row says why.",
      "rows": [
        {
          "metric": "Resolved rate on the same 33 SWE-bench Verified instances",
          "aValue": 0.8485,
          "bValue": 0.7273,
          "unit": "rate",
          "aDisplay": "85% (28/33)",
          "bDisplay": "73% (24/33)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (GPT 5.2 69% to 93%; DeepSeek V3.2 56% to 85%), so this sample cannot separate them.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-same-instance-leaderboard",
          "aN": 33,
          "bN": 33,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.6908,
            0.9335
          ],
          "bRange": [
            0.5578,
            0.8493
          ]
        },
        {
          "metric": "Model calls per instance",
          "aValue": 35.6,
          "bValue": 88.2,
          "unit": "calls",
          "aDisplay": "35.6",
          "bDisplay": "88.2",
          "winner": "unclear",
          "basis": "More or fewer calls is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-model-calls",
          "aN": 33,
          "bN": 33,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances"
        },
        {
          "metric": "Recorded cost per resolved instance: Agent vs the public panel",
          "aValue": 0.628,
          "bValue": 0.637,
          "unit": "usd",
          "aDisplay": "$0.63",
          "bDisplay": "$0.64",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.63 vs $0.64) is not tested against run-to-run variation.",
          "studySlug": "cost-thought-experiments",
          "chartId": "cost-per-resolved-agent-vs-panel",
          "aN": 28,
          "bN": 24,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances"
        }
      ]
    },
    {
      "slug": "gpt-5-2-vs-gemini-3-flash",
      "a": "gpt-5-2",
      "b": "gemini-3-flash",
      "title": "GPT 5.2 vs Gemini 3 Flash",
      "seoTitle": "GPT 5.2 vs Gemini 3 Flash: measured benchmarks",
      "description": "GPT 5.2 vs Gemini 3 Flash: 3 measured metrics from 2 studies (Resolved rate on the same 33 SWE-bench Verified instances; more), with sample sizes and intervals.",
      "verdict": "GPT 5.2 and Gemini 3 Flash share 3 measured metrics from 2 studies. No row separates them: every interval or run range overlaps, too few runs were recorded, no interval was recorded, or more is not better for that metric. The rows are 1 tie and 2 unclear; each row says why.",
      "rows": [
        {
          "metric": "Resolved rate on the same 33 SWE-bench Verified instances",
          "aValue": 0.8485,
          "bValue": 0.8182,
          "unit": "rate",
          "aDisplay": "85% (28/33)",
          "bDisplay": "82% (27/33)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (GPT 5.2 69% to 93%; Gemini 3 Flash 66% to 91%), so this sample cannot separate them.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-same-instance-leaderboard",
          "aN": 33,
          "bN": 33,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.6908,
            0.9335
          ],
          "bRange": [
            0.6561,
            0.9139
          ]
        },
        {
          "metric": "Model calls per instance",
          "aValue": 35.6,
          "bValue": 54.2,
          "unit": "calls",
          "aDisplay": "35.6",
          "bDisplay": "54.2",
          "winner": "unclear",
          "basis": "More or fewer calls is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-model-calls",
          "aN": 33,
          "bN": 33,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances"
        },
        {
          "metric": "Recorded cost per resolved instance: Agent vs the public panel",
          "aValue": 0.628,
          "bValue": 0.436,
          "unit": "usd",
          "aDisplay": "$0.63",
          "bDisplay": "$0.44",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.63 vs $0.44) is not tested against run-to-run variation.",
          "studySlug": "cost-thought-experiments",
          "chartId": "cost-per-resolved-agent-vs-panel",
          "aN": 28,
          "bN": 27,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances"
        }
      ]
    },
    {
      "slug": "gpt-5-2-vs-glm-5",
      "a": "gpt-5-2",
      "b": "glm-5",
      "title": "GPT 5.2 vs GLM 5",
      "seoTitle": "GPT 5.2 vs GLM 5: measured benchmarks",
      "description": "GPT 5.2 vs GLM 5: 3 measured metrics from 2 studies (Resolved rate on the same 33 SWE-bench Verified instances; more), with sample sizes and intervals.",
      "verdict": "GPT 5.2 and GLM 5 share 3 measured metrics from 2 studies. No row separates them: every interval or run range overlaps, too few runs were recorded, no interval was recorded, or more is not better for that metric. The rows are 1 tie and 2 unclear; each row says why.",
      "rows": [
        {
          "metric": "Resolved rate on the same 33 SWE-bench Verified instances",
          "aValue": 0.8485,
          "bValue": 0.7879,
          "unit": "rate",
          "aDisplay": "85% (28/33)",
          "bDisplay": "79% (26/33)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (GPT 5.2 69% to 93%; GLM 5 62% to 89%), so this sample cannot separate them.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-same-instance-leaderboard",
          "aN": 33,
          "bN": 33,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.6908,
            0.9335
          ],
          "bRange": [
            0.6225,
            0.8932
          ]
        },
        {
          "metric": "Model calls per instance",
          "aValue": 35.6,
          "bValue": 77.5,
          "unit": "calls",
          "aDisplay": "35.6",
          "bDisplay": "77.5",
          "winner": "unclear",
          "basis": "More or fewer calls is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-model-calls",
          "aN": 33,
          "bN": 33,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances"
        },
        {
          "metric": "Recorded cost per resolved instance: Agent vs the public panel",
          "aValue": 0.628,
          "bValue": 0.667,
          "unit": "usd",
          "aDisplay": "$0.63",
          "bDisplay": "$0.67",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.63 vs $0.67) is not tested against run-to-run variation.",
          "studySlug": "cost-thought-experiments",
          "chartId": "cost-per-resolved-agent-vs-panel",
          "aN": 28,
          "bN": 26,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances"
        }
      ]
    },
    {
      "slug": "gpt-5-2-vs-gpt-5-mini",
      "a": "gpt-5-2",
      "b": "gpt-5-mini",
      "title": "GPT 5.2 vs GPT 5 mini",
      "seoTitle": "GPT 5.2 vs GPT 5 mini: measured benchmarks",
      "description": "GPT 5.2 vs GPT 5 mini: 3 measured metrics from 2 studies (Resolved rate on the same 33 SWE-bench Verified instances; more), with sample sizes and intervals.",
      "verdict": "GPT 5.2 and GPT 5 mini share 3 measured metrics from 2 studies. No row separates them: every interval or run range overlaps, too few runs were recorded, no interval was recorded, or more is not better for that metric. The rows are 1 tie and 2 unclear; each row says why.",
      "rows": [
        {
          "metric": "Resolved rate on the same 33 SWE-bench Verified instances",
          "aValue": 0.8485,
          "bValue": 0.6364,
          "unit": "rate",
          "aDisplay": "85% (28/33)",
          "bDisplay": "64% (21/33)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (GPT 5.2 69% to 93%; GPT 5 mini 47% to 78%), so this sample cannot separate them.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-same-instance-leaderboard",
          "aN": 33,
          "bN": 33,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "public mini-SWE-agent v2 run, same instances",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.6908,
            0.9335
          ],
          "bRange": [
            0.4662,
            0.7781
          ]
        },
        {
          "metric": "Model calls per instance",
          "aValue": 35.6,
          "bValue": 20.8,
          "unit": "calls",
          "aDisplay": "35.6",
          "bDisplay": "20.8",
          "winner": "unclear",
          "basis": "More or fewer calls is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-model-calls",
          "aN": 33,
          "bN": 33,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "public mini-SWE-agent v2 run, same instances"
        },
        {
          "metric": "Recorded cost per resolved instance: Agent vs the public panel",
          "aValue": 0.628,
          "bValue": 0.08,
          "unit": "usd",
          "aDisplay": "$0.63",
          "bDisplay": "$0.080",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.63 vs $0.080, 7.8x) is not tested against run-to-run variation.",
          "studySlug": "cost-thought-experiments",
          "chartId": "cost-per-resolved-agent-vs-panel",
          "aN": 28,
          "bN": 21,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "public mini-SWE-agent v2 run, same instances"
        }
      ]
    },
    {
      "slug": "gpt-5-2-vs-kimi-k2-5",
      "a": "gpt-5-2",
      "b": "kimi-k2-5",
      "title": "GPT 5.2 vs Kimi K2.5",
      "seoTitle": "GPT 5.2 vs Kimi K2.5: measured benchmarks",
      "description": "GPT 5.2 vs Kimi K2.5: 3 measured metrics from 2 studies (Resolved rate on the same 33 SWE-bench Verified instances; more), with sample sizes and intervals.",
      "verdict": "GPT 5.2 and Kimi K2.5 share 3 measured metrics from 2 studies. No row separates them: every interval or run range overlaps, too few runs were recorded, no interval was recorded, or more is not better for that metric. The rows are 1 tie and 2 unclear; each row says why.",
      "rows": [
        {
          "metric": "Resolved rate on the same 33 SWE-bench Verified instances",
          "aValue": 0.8485,
          "bValue": 0.697,
          "unit": "rate",
          "aDisplay": "85% (28/33)",
          "bDisplay": "70% (23/33)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (GPT 5.2 69% to 93%; Kimi K2.5 53% to 83%), so this sample cannot separate them.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-same-instance-leaderboard",
          "aN": 33,
          "bN": 33,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.6908,
            0.9335
          ],
          "bRange": [
            0.5266,
            0.8262
          ]
        },
        {
          "metric": "Model calls per instance",
          "aValue": 35.6,
          "bValue": 56.7,
          "unit": "calls",
          "aDisplay": "35.6",
          "bDisplay": "56.7",
          "winner": "unclear",
          "basis": "More or fewer calls is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-model-calls",
          "aN": 33,
          "bN": 33,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances"
        },
        {
          "metric": "Recorded cost per resolved instance: Agent vs the public panel",
          "aValue": 0.628,
          "bValue": 0.256,
          "unit": "usd",
          "aDisplay": "$0.63",
          "bDisplay": "$0.26",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.63 vs $0.26, 2.5x) is not tested against run-to-run variation.",
          "studySlug": "cost-thought-experiments",
          "chartId": "cost-per-resolved-agent-vs-panel",
          "aN": 28,
          "bN": 23,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances"
        }
      ]
    },
    {
      "slug": "gpt-5-2-vs-minimax-m2-5",
      "a": "gpt-5-2",
      "b": "minimax-m2-5",
      "title": "GPT 5.2 vs MiniMax M2.5",
      "seoTitle": "GPT 5.2 vs MiniMax M2.5: measured benchmarks",
      "description": "GPT 5.2 vs MiniMax M2.5: 3 measured metrics from 2 studies (Resolved rate on the same 33 SWE-bench Verified instances; more), with sample sizes and intervals.",
      "verdict": "GPT 5.2 and MiniMax M2.5 share 3 measured metrics from 2 studies. No row separates them: every interval or run range overlaps, too few runs were recorded, no interval was recorded, or more is not better for that metric. The rows are 1 tie and 2 unclear; each row says why.",
      "rows": [
        {
          "metric": "Resolved rate on the same 33 SWE-bench Verified instances",
          "aValue": 0.8485,
          "bValue": 0.697,
          "unit": "rate",
          "aDisplay": "85% (28/33)",
          "bDisplay": "70% (23/33)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (GPT 5.2 69% to 93%; MiniMax M2.5 53% to 83%), so this sample cannot separate them.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-same-instance-leaderboard",
          "aN": 33,
          "bN": 33,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.6908,
            0.9335
          ],
          "bRange": [
            0.5266,
            0.8262
          ]
        },
        {
          "metric": "Model calls per instance",
          "aValue": 35.6,
          "bValue": 58.4,
          "unit": "calls",
          "aDisplay": "35.6",
          "bDisplay": "58.4",
          "winner": "unclear",
          "basis": "More or fewer calls is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-model-calls",
          "aN": 33,
          "bN": 33,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances"
        },
        {
          "metric": "Recorded cost per resolved instance: Agent vs the public panel",
          "aValue": 0.628,
          "bValue": 0.107,
          "unit": "usd",
          "aDisplay": "$0.63",
          "bDisplay": "$0.11",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.63 vs $0.11, 5.9x) is not tested against run-to-run variation.",
          "studySlug": "cost-thought-experiments",
          "chartId": "cost-per-resolved-agent-vs-panel",
          "aN": 28,
          "bN": 23,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances"
        }
      ]
    },
    {
      "slug": "groq-vs-baseten",
      "a": "groq",
      "b": "baseten",
      "title": "Groq vs Baseten",
      "seoTitle": "Groq vs Baseten: price per model",
      "description": "Groq vs Baseten: 3 list prices per million tokens for the same models, as reported by OpenRouter’s public API, with the gap stated.",
      "verdict": "Groq and Baseten share 3 reported list prices from one study. No row names a winner: a reported list price has no interval, so each gap is stated, not ranked. Prices change often; the snapshot date is in each row’s context. The rows are 3 unclear; each row says why.",
      "rows": [
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Input)",
          "aValue": 0.15,
          "bValue": 0.1,
          "unit": "usd",
          "aDisplay": "$0.15",
          "bDisplay": "$0.10",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.15 vs $0.10, 1.5x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Output)",
          "aValue": 0.6,
          "bValue": 0.5,
          "unit": "usd",
          "aDisplay": "$0.60",
          "bDisplay": "$0.50",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.60 vs $0.50, 1.2x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Cache read)",
          "aValue": 0.075,
          "bValue": 0.1,
          "unit": "usd",
          "aDisplay": "$0.075",
          "bDisplay": "$0.10",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.075 vs $0.10, 1.3x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "groq-vs-siliconflow",
      "a": "groq",
      "b": "siliconflow",
      "title": "Groq vs SiliconFlow",
      "seoTitle": "Groq vs SiliconFlow: price per model",
      "description": "Groq vs SiliconFlow: 3 list prices per million tokens for the same models, as reported by OpenRouter’s public API, with the gap stated.",
      "verdict": "Groq and SiliconFlow share 3 reported list prices from one study. No row names a winner: a reported list price has no interval, so each gap is stated, not ranked. Prices change often; the snapshot date is in each row’s context. The rows are 3 ties; each row says why.",
      "rows": [
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Input)",
          "aValue": 0.15,
          "bValue": 0.15,
          "unit": "usd",
          "aDisplay": "$0.15",
          "bDisplay": "$0.15",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Output)",
          "aValue": 0.6,
          "bValue": 0.6,
          "unit": "usd",
          "aDisplay": "$0.60",
          "bDisplay": "$0.60",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Cache read)",
          "aValue": 0.075,
          "bValue": 0.075,
          "unit": "usd",
          "aDisplay": "$0.075",
          "bDisplay": "$0.075",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "kimi-k2-5-vs-gpt-5-mini",
      "a": "kimi-k2-5",
      "b": "gpt-5-mini",
      "title": "Kimi K2.5 vs GPT 5 mini",
      "seoTitle": "Kimi K2.5 vs GPT 5 mini: measured benchmarks",
      "description": "Kimi K2.5 vs GPT 5 mini: 3 measured metrics from 2 studies (Resolved rate on the same 33 SWE-bench Verified instances; more), with sample sizes and intervals.",
      "verdict": "Kimi K2.5 and GPT 5 mini share 3 measured metrics from 2 studies. No row separates them: every interval or run range overlaps, too few runs were recorded, no interval was recorded, or more is not better for that metric. The rows are 1 tie and 2 unclear; each row says why.",
      "rows": [
        {
          "metric": "Resolved rate on the same 33 SWE-bench Verified instances",
          "aValue": 0.697,
          "bValue": 0.6364,
          "unit": "rate",
          "aDisplay": "70% (23/33)",
          "bDisplay": "64% (21/33)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Kimi K2.5 53% to 83%; GPT 5 mini 47% to 78%), so this sample cannot separate them.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-same-instance-leaderboard",
          "aN": 33,
          "bN": 33,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "public mini-SWE-agent v2 run, same instances",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.5266,
            0.8262
          ],
          "bRange": [
            0.4662,
            0.7781
          ]
        },
        {
          "metric": "Model calls per instance",
          "aValue": 56.7,
          "bValue": 20.8,
          "unit": "calls",
          "aDisplay": "56.7",
          "bDisplay": "20.8",
          "winner": "unclear",
          "basis": "More or fewer calls is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-model-calls",
          "aN": 33,
          "bN": 33,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "public mini-SWE-agent v2 run, same instances"
        },
        {
          "metric": "Recorded cost per resolved instance: Agent vs the public panel",
          "aValue": 0.256,
          "bValue": 0.08,
          "unit": "usd",
          "aDisplay": "$0.26",
          "bDisplay": "$0.080",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.26 vs $0.080, 3.2x) is not tested against run-to-run variation.",
          "studySlug": "cost-thought-experiments",
          "chartId": "cost-per-resolved-agent-vs-panel",
          "aN": 23,
          "bN": 21,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "public mini-SWE-agent v2 run, same instances"
        }
      ]
    },
    {
      "slug": "minimax-m2-5-vs-gpt-5-mini",
      "a": "minimax-m2-5",
      "b": "gpt-5-mini",
      "title": "MiniMax M2.5 vs GPT 5 mini",
      "seoTitle": "MiniMax M2.5 vs GPT 5 mini: measured benchmarks",
      "description": "MiniMax M2.5 vs GPT 5 mini: 3 measured metrics from 2 studies, with sample sizes, intervals and every failure counted.",
      "verdict": "MiniMax M2.5 and GPT 5 mini share 3 measured metrics from 2 studies. No row separates them: every interval or run range overlaps, too few runs were recorded, no interval was recorded, or more is not better for that metric. The rows are 1 tie and 2 unclear; each row says why.",
      "rows": [
        {
          "metric": "Resolved rate on the same 33 SWE-bench Verified instances",
          "aValue": 0.697,
          "bValue": 0.6364,
          "unit": "rate",
          "aDisplay": "70% (23/33)",
          "bDisplay": "64% (21/33)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (MiniMax M2.5 53% to 83%; GPT 5 mini 47% to 78%), so this sample cannot separate them.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-same-instance-leaderboard",
          "aN": 33,
          "bN": 33,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "public mini-SWE-agent v2 run, same instances",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.5266,
            0.8262
          ],
          "bRange": [
            0.4662,
            0.7781
          ]
        },
        {
          "metric": "Model calls per instance",
          "aValue": 58.4,
          "bValue": 20.8,
          "unit": "calls",
          "aDisplay": "58.4",
          "bDisplay": "20.8",
          "winner": "unclear",
          "basis": "More or fewer calls is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-model-calls",
          "aN": 33,
          "bN": 33,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "public mini-SWE-agent v2 run, same instances"
        },
        {
          "metric": "Recorded cost per resolved instance: Agent vs the public panel",
          "aValue": 0.107,
          "bValue": 0.08,
          "unit": "usd",
          "aDisplay": "$0.11",
          "bDisplay": "$0.080",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.11 vs $0.080) is not tested against run-to-run variation.",
          "studySlug": "cost-thought-experiments",
          "chartId": "cost-per-resolved-agent-vs-panel",
          "aN": 23,
          "bN": 21,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "public mini-SWE-agent v2 run, same instances"
        }
      ]
    },
    {
      "slug": "minimax-m2-5-vs-kimi-k2-5",
      "a": "minimax-m2-5",
      "b": "kimi-k2-5",
      "title": "MiniMax M2.5 vs Kimi K2.5",
      "seoTitle": "MiniMax M2.5 vs Kimi K2.5: measured benchmarks",
      "description": "MiniMax M2.5 vs Kimi K2.5: 3 measured metrics from 2 studies (Resolved rate on the same 33 SWE-bench Verified instances; more), with sample sizes and intervals.",
      "verdict": "MiniMax M2.5 and Kimi K2.5 share 3 measured metrics from 2 studies. No row separates them: every interval or run range overlaps, too few runs were recorded, no interval was recorded, or more is not better for that metric. The rows are 1 tie and 2 unclear; each row says why.",
      "rows": [
        {
          "metric": "Resolved rate on the same 33 SWE-bench Verified instances",
          "aValue": 0.697,
          "bValue": 0.697,
          "unit": "rate",
          "aDisplay": "70% (23/33)",
          "bDisplay": "70% (23/33)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (MiniMax M2.5 53% to 83%; Kimi K2.5 53% to 83%), so this sample cannot separate them.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-same-instance-leaderboard",
          "aN": 33,
          "bN": 33,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.5266,
            0.8262
          ],
          "bRange": [
            0.5266,
            0.8262
          ]
        },
        {
          "metric": "Model calls per instance",
          "aValue": 58.4,
          "bValue": 56.7,
          "unit": "calls",
          "aDisplay": "58.4",
          "bDisplay": "56.7",
          "winner": "unclear",
          "basis": "More or fewer calls is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-model-calls",
          "aN": 33,
          "bN": 33,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances"
        },
        {
          "metric": "Recorded cost per resolved instance: Agent vs the public panel",
          "aValue": 0.107,
          "bValue": 0.256,
          "unit": "usd",
          "aDisplay": "$0.11",
          "bDisplay": "$0.26",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.11 vs $0.26, 2.4x) is not tested against run-to-run variation.",
          "studySlug": "cost-thought-experiments",
          "n": 23,
          "chartId": "cost-per-resolved-agent-vs-panel",
          "aN": 23,
          "bN": 23,
          "aContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances"
        }
      ]
    },
    {
      "slug": "agent-harness-vs-claude-haiku-4-5",
      "a": "agent-harness",
      "b": "claude-haiku-4-5",
      "title": "Agent vs Claude Haiku 4.5",
      "seoTitle": "Agent vs Claude Haiku 4.5: measured benchmarks",
      "description": "Agent vs Claude Haiku 4.5: 2 measured metrics from one study (Resolved rate on the same 33 SWE-bench Verified instances; more), with sample sizes and intervals.",
      "verdict": "Agent and Claude Haiku 4.5 share 2 measured metrics and 1 list-price calculation from 2 studies. No row separates them: every interval or run range overlaps, too few runs were recorded, no interval was recorded, or more is not better for that metric. The rows are 1 tie and 2 unclear; each row says why. Calculation rows are derived from list prices and recorded counts; they are not bills or runs. Agent is a full pipeline on one model; Claude Haiku 4.5 ran under a different, simpler harness, so this compares systems, not models.",
      "rows": [
        {
          "metric": "Resolved rate on the same 33 SWE-bench Verified instances",
          "aValue": 0.7576,
          "bValue": 0.7576,
          "unit": "rate",
          "aDisplay": "76% (25/33)",
          "bDisplay": "76% (25/33)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Agent 59% to 87%; Claude Haiku 4.5 59% to 87%), so this sample cannot separate them.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-same-instance-leaderboard",
          "aN": 33,
          "bN": 33,
          "aContext": "full pipeline on Claude Sonnet 5.5",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.5898,
            0.8717
          ],
          "bRange": [
            0.5898,
            0.8717
          ]
        },
        {
          "metric": "Model calls per instance",
          "aValue": 49.5,
          "bValue": 68.5,
          "unit": "calls",
          "aDisplay": "49.5",
          "bDisplay": "68.5",
          "winner": "unclear",
          "basis": "More or fewer calls is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-model-calls",
          "aN": 33,
          "bN": 33,
          "aContext": "full pipeline on Claude Sonnet 5.5",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances"
        },
        {
          "metric": "Recorded cost per resolved instance: Agent vs the public panel",
          "aValue": 3.706,
          "bValue": 0.479,
          "unit": "usd",
          "aDisplay": "$3.71",
          "bDisplay": "$0.48",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($3.71 vs $0.48, 7.7x) is not tested against run-to-run variation.",
          "studySlug": "cost-thought-experiments",
          "n": 25,
          "chartId": "cost-per-resolved-agent-vs-panel",
          "aN": 25,
          "bN": 25,
          "aContext": "full pipeline on Claude Sonnet 5.5 · notional",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "calculation": true
        }
      ]
    },
    {
      "slug": "agent-harness-vs-claude-opus-4-5",
      "a": "agent-harness",
      "b": "claude-opus-4-5",
      "title": "Agent vs Claude Opus 4.5",
      "seoTitle": "Agent vs Claude Opus 4.5: measured benchmarks",
      "description": "Agent vs Claude Opus 4.5: 2 measured metrics from one study (Resolved rate on the same 33 SWE-bench Verified instances; more), with sample sizes and intervals.",
      "verdict": "Agent and Claude Opus 4.5 share 2 measured metrics and 1 list-price calculation from 2 studies. No row separates them: every interval or run range overlaps, too few runs were recorded, no interval was recorded, or more is not better for that metric. The rows are 1 tie and 2 unclear; each row says why. Calculation rows are derived from list prices and recorded counts; they are not bills or runs. Agent is a full pipeline on one model; Claude Opus 4.5 ran under a different, simpler harness, so this compares systems, not models.",
      "rows": [
        {
          "metric": "Resolved rate on the same 33 SWE-bench Verified instances",
          "aValue": 0.7576,
          "bValue": 0.7273,
          "unit": "rate",
          "aDisplay": "76% (25/33)",
          "bDisplay": "73% (24/33)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Agent 59% to 87%; Claude Opus 4.5 56% to 85%), so this sample cannot separate them.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-same-instance-leaderboard",
          "aN": 33,
          "bN": 33,
          "aContext": "full pipeline on Claude Sonnet 5.5",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.5898,
            0.8717
          ],
          "bRange": [
            0.5578,
            0.8493
          ]
        },
        {
          "metric": "Model calls per instance",
          "aValue": 49.5,
          "bValue": 35.9,
          "unit": "calls",
          "aDisplay": "49.5",
          "bDisplay": "35.9",
          "winner": "unclear",
          "basis": "More or fewer calls is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-model-calls",
          "aN": 33,
          "bN": 33,
          "aContext": "full pipeline on Claude Sonnet 5.5",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances"
        },
        {
          "metric": "Recorded cost per resolved instance: Agent vs the public panel",
          "aValue": 3.706,
          "bValue": 1.184,
          "unit": "usd",
          "aDisplay": "$3.71",
          "bDisplay": "$1.18",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($3.71 vs $1.18, 3.1x) is not tested against run-to-run variation.",
          "studySlug": "cost-thought-experiments",
          "chartId": "cost-per-resolved-agent-vs-panel",
          "aN": 25,
          "bN": 24,
          "aContext": "full pipeline on Claude Sonnet 5.5 · notional",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "calculation": true
        }
      ]
    },
    {
      "slug": "agent-harness-vs-claude-opus-4-6",
      "a": "agent-harness",
      "b": "claude-opus-4-6",
      "title": "Agent vs Claude Opus 4.6",
      "seoTitle": "Agent vs Claude Opus 4.6: measured benchmarks",
      "description": "Agent vs Claude Opus 4.6: 2 measured metrics from one study (Resolved rate on the same 33 SWE-bench Verified instances; more), with sample sizes and intervals.",
      "verdict": "Agent and Claude Opus 4.6 share 2 measured metrics and 1 list-price calculation from 2 studies. No row separates them: every interval or run range overlaps, too few runs were recorded, no interval was recorded, or more is not better for that metric. The rows are 1 tie and 2 unclear; each row says why. Calculation rows are derived from list prices and recorded counts; they are not bills or runs. Agent is a full pipeline on one model; Claude Opus 4.6 ran under a different, simpler harness, so this compares systems, not models.",
      "rows": [
        {
          "metric": "Resolved rate on the same 33 SWE-bench Verified instances",
          "aValue": 0.7576,
          "bValue": 0.697,
          "unit": "rate",
          "aDisplay": "76% (25/33)",
          "bDisplay": "70% (23/33)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Agent 59% to 87%; Claude Opus 4.6 53% to 83%), so this sample cannot separate them.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-same-instance-leaderboard",
          "aN": 33,
          "bN": 33,
          "aContext": "full pipeline on Claude Sonnet 5.5",
          "bContext": "public mini-SWE-agent v2 run, same instances",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.5898,
            0.8717
          ],
          "bRange": [
            0.5266,
            0.8262
          ]
        },
        {
          "metric": "Model calls per instance",
          "aValue": 49.5,
          "bValue": 28.9,
          "unit": "calls",
          "aDisplay": "49.5",
          "bDisplay": "28.9",
          "winner": "unclear",
          "basis": "More or fewer calls is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-model-calls",
          "aN": 33,
          "bN": 33,
          "aContext": "full pipeline on Claude Sonnet 5.5",
          "bContext": "public mini-SWE-agent v2 run, same instances"
        },
        {
          "metric": "Recorded cost per resolved instance: Agent vs the public panel",
          "aValue": 3.706,
          "bValue": 0.875,
          "unit": "usd",
          "aDisplay": "$3.71",
          "bDisplay": "$0.88",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($3.71 vs $0.88, 4.2x) is not tested against run-to-run variation.",
          "studySlug": "cost-thought-experiments",
          "chartId": "cost-per-resolved-agent-vs-panel",
          "aN": 25,
          "bN": 23,
          "aContext": "full pipeline on Claude Sonnet 5.5 · notional",
          "bContext": "public mini-SWE-agent v2 run, same instances",
          "calculation": true
        }
      ]
    },
    {
      "slug": "agent-harness-vs-claude-sonnet-4-5",
      "a": "agent-harness",
      "b": "claude-sonnet-4-5",
      "title": "Agent vs Claude Sonnet 4.5",
      "seoTitle": "Agent vs Claude Sonnet 4.5: measured benchmarks",
      "description": "Agent vs Claude Sonnet 4.5: 2 measured metrics from one study, with sample sizes, intervals and every failure counted.",
      "verdict": "Agent and Claude Sonnet 4.5 share 2 measured metrics and 1 list-price calculation from 2 studies. No row separates them: every interval or run range overlaps, too few runs were recorded, no interval was recorded, or more is not better for that metric. The rows are 1 tie and 2 unclear; each row says why. Calculation rows are derived from list prices and recorded counts; they are not bills or runs. Agent is a full pipeline on one model; Claude Sonnet 4.5 ran under a different, simpler harness, so this compares systems, not models.",
      "rows": [
        {
          "metric": "Resolved rate on the same 33 SWE-bench Verified instances",
          "aValue": 0.7576,
          "bValue": 0.7576,
          "unit": "rate",
          "aDisplay": "76% (25/33)",
          "bDisplay": "76% (25/33)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Agent 59% to 87%; Claude Sonnet 4.5 59% to 87%), so this sample cannot separate them.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-same-instance-leaderboard",
          "aN": 33,
          "bN": 33,
          "aContext": "full pipeline on Claude Sonnet 5.5",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.5898,
            0.8717
          ],
          "bRange": [
            0.5898,
            0.8717
          ]
        },
        {
          "metric": "Model calls per instance",
          "aValue": 49.5,
          "bValue": 51,
          "unit": "calls",
          "aDisplay": "49.5",
          "bDisplay": "51",
          "winner": "unclear",
          "basis": "More or fewer calls is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-model-calls",
          "aN": 33,
          "bN": 33,
          "aContext": "full pipeline on Claude Sonnet 5.5",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances"
        },
        {
          "metric": "Recorded cost per resolved instance: Agent vs the public panel",
          "aValue": 3.706,
          "bValue": 0.913,
          "unit": "usd",
          "aDisplay": "$3.71",
          "bDisplay": "$0.91",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($3.71 vs $0.91, 4.1x) is not tested against run-to-run variation.",
          "studySlug": "cost-thought-experiments",
          "n": 25,
          "chartId": "cost-per-resolved-agent-vs-panel",
          "aN": 25,
          "bN": 25,
          "aContext": "full pipeline on Claude Sonnet 5.5 · notional",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "calculation": true
        }
      ]
    },
    {
      "slug": "agent-harness-vs-deepseek-v3-2",
      "a": "agent-harness",
      "b": "deepseek-v3-2",
      "title": "Agent vs DeepSeek V3.2",
      "seoTitle": "Agent vs DeepSeek V3.2: measured benchmarks",
      "description": "Agent vs DeepSeek V3.2: 2 measured metrics from one study (Resolved rate on the same 33 SWE-bench Verified instances; more), with sample sizes and intervals.",
      "verdict": "Agent and DeepSeek V3.2 share 2 measured metrics and 1 list-price calculation from 2 studies. No row separates them: every interval or run range overlaps, too few runs were recorded, no interval was recorded, or more is not better for that metric. The rows are 1 tie and 2 unclear; each row says why. Calculation rows are derived from list prices and recorded counts; they are not bills or runs. Agent is a full pipeline on one model; DeepSeek V3.2 ran under a different, simpler harness, so this compares systems, not models.",
      "rows": [
        {
          "metric": "Resolved rate on the same 33 SWE-bench Verified instances",
          "aValue": 0.7576,
          "bValue": 0.7273,
          "unit": "rate",
          "aDisplay": "76% (25/33)",
          "bDisplay": "73% (24/33)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Agent 59% to 87%; DeepSeek V3.2 56% to 85%), so this sample cannot separate them.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-same-instance-leaderboard",
          "aN": 33,
          "bN": 33,
          "aContext": "full pipeline on Claude Sonnet 5.5",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.5898,
            0.8717
          ],
          "bRange": [
            0.5578,
            0.8493
          ]
        },
        {
          "metric": "Model calls per instance",
          "aValue": 49.5,
          "bValue": 88.2,
          "unit": "calls",
          "aDisplay": "49.5",
          "bDisplay": "88.2",
          "winner": "unclear",
          "basis": "More or fewer calls is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-model-calls",
          "aN": 33,
          "bN": 33,
          "aContext": "full pipeline on Claude Sonnet 5.5",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances"
        },
        {
          "metric": "Recorded cost per resolved instance: Agent vs the public panel",
          "aValue": 3.706,
          "bValue": 0.637,
          "unit": "usd",
          "aDisplay": "$3.71",
          "bDisplay": "$0.64",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($3.71 vs $0.64, 5.8x) is not tested against run-to-run variation.",
          "studySlug": "cost-thought-experiments",
          "chartId": "cost-per-resolved-agent-vs-panel",
          "aN": 25,
          "bN": 24,
          "aContext": "full pipeline on Claude Sonnet 5.5 · notional",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "calculation": true
        }
      ]
    },
    {
      "slug": "agent-harness-vs-gemini-3-flash",
      "a": "agent-harness",
      "b": "gemini-3-flash",
      "title": "Agent vs Gemini 3 Flash",
      "seoTitle": "Agent vs Gemini 3 Flash: measured benchmarks",
      "description": "Agent vs Gemini 3 Flash: 2 measured metrics from one study (Resolved rate on the same 33 SWE-bench Verified instances; more), with sample sizes and intervals.",
      "verdict": "Agent and Gemini 3 Flash share 2 measured metrics and 1 list-price calculation from 2 studies. No row separates them: every interval or run range overlaps, too few runs were recorded, no interval was recorded, or more is not better for that metric. The rows are 1 tie and 2 unclear; each row says why. Calculation rows are derived from list prices and recorded counts; they are not bills or runs. Agent is a full pipeline on one model; Gemini 3 Flash ran under a different, simpler harness, so this compares systems, not models.",
      "rows": [
        {
          "metric": "Resolved rate on the same 33 SWE-bench Verified instances",
          "aValue": 0.7576,
          "bValue": 0.8182,
          "unit": "rate",
          "aDisplay": "76% (25/33)",
          "bDisplay": "82% (27/33)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Agent 59% to 87%; Gemini 3 Flash 66% to 91%), so this sample cannot separate them.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-same-instance-leaderboard",
          "aN": 33,
          "bN": 33,
          "aContext": "full pipeline on Claude Sonnet 5.5",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.5898,
            0.8717
          ],
          "bRange": [
            0.6561,
            0.9139
          ]
        },
        {
          "metric": "Model calls per instance",
          "aValue": 49.5,
          "bValue": 54.2,
          "unit": "calls",
          "aDisplay": "49.5",
          "bDisplay": "54.2",
          "winner": "unclear",
          "basis": "More or fewer calls is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-model-calls",
          "aN": 33,
          "bN": 33,
          "aContext": "full pipeline on Claude Sonnet 5.5",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances"
        },
        {
          "metric": "Recorded cost per resolved instance: Agent vs the public panel",
          "aValue": 3.706,
          "bValue": 0.436,
          "unit": "usd",
          "aDisplay": "$3.71",
          "bDisplay": "$0.44",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($3.71 vs $0.44, 8.5x) is not tested against run-to-run variation.",
          "studySlug": "cost-thought-experiments",
          "chartId": "cost-per-resolved-agent-vs-panel",
          "aN": 25,
          "bN": 27,
          "aContext": "full pipeline on Claude Sonnet 5.5 · notional",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "calculation": true
        }
      ]
    },
    {
      "slug": "agent-harness-vs-glm-5",
      "a": "agent-harness",
      "b": "glm-5",
      "title": "Agent vs GLM 5",
      "seoTitle": "Agent vs GLM 5: measured benchmarks",
      "description": "Agent vs GLM 5: 2 measured metrics from one study (Resolved rate on the same 33 SWE-bench Verified instances; more), with sample sizes and intervals.",
      "verdict": "Agent and GLM 5 share 2 measured metrics and 1 list-price calculation from 2 studies. No row separates them: every interval or run range overlaps, too few runs were recorded, no interval was recorded, or more is not better for that metric. The rows are 1 tie and 2 unclear; each row says why. Calculation rows are derived from list prices and recorded counts; they are not bills or runs. Agent is a full pipeline on one model; GLM 5 ran under a different, simpler harness, so this compares systems, not models.",
      "rows": [
        {
          "metric": "Resolved rate on the same 33 SWE-bench Verified instances",
          "aValue": 0.7576,
          "bValue": 0.7879,
          "unit": "rate",
          "aDisplay": "76% (25/33)",
          "bDisplay": "79% (26/33)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Agent 59% to 87%; GLM 5 62% to 89%), so this sample cannot separate them.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-same-instance-leaderboard",
          "aN": 33,
          "bN": 33,
          "aContext": "full pipeline on Claude Sonnet 5.5",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.5898,
            0.8717
          ],
          "bRange": [
            0.6225,
            0.8932
          ]
        },
        {
          "metric": "Model calls per instance",
          "aValue": 49.5,
          "bValue": 77.5,
          "unit": "calls",
          "aDisplay": "49.5",
          "bDisplay": "77.5",
          "winner": "unclear",
          "basis": "More or fewer calls is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-model-calls",
          "aN": 33,
          "bN": 33,
          "aContext": "full pipeline on Claude Sonnet 5.5",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances"
        },
        {
          "metric": "Recorded cost per resolved instance: Agent vs the public panel",
          "aValue": 3.706,
          "bValue": 0.667,
          "unit": "usd",
          "aDisplay": "$3.71",
          "bDisplay": "$0.67",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($3.71 vs $0.67, 5.6x) is not tested against run-to-run variation.",
          "studySlug": "cost-thought-experiments",
          "chartId": "cost-per-resolved-agent-vs-panel",
          "aN": 25,
          "bN": 26,
          "aContext": "full pipeline on Claude Sonnet 5.5 · notional",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "calculation": true
        }
      ]
    },
    {
      "slug": "agent-harness-vs-gpt-5-2",
      "a": "agent-harness",
      "b": "gpt-5-2",
      "title": "Agent vs GPT 5.2",
      "seoTitle": "Agent vs GPT 5.2: measured benchmarks",
      "description": "Agent vs GPT 5.2: 2 measured metrics from one study (Resolved rate on the same 33 SWE-bench Verified instances; more), with sample sizes and intervals.",
      "verdict": "Agent and GPT 5.2 share 2 measured metrics and 1 list-price calculation from 2 studies. No row separates them: every interval or run range overlaps, too few runs were recorded, no interval was recorded, or more is not better for that metric. The rows are 1 tie and 2 unclear; each row says why. Calculation rows are derived from list prices and recorded counts; they are not bills or runs. Agent is a full pipeline on one model; GPT 5.2 ran under a different, simpler harness, so this compares systems, not models.",
      "rows": [
        {
          "metric": "Resolved rate on the same 33 SWE-bench Verified instances",
          "aValue": 0.7576,
          "bValue": 0.8485,
          "unit": "rate",
          "aDisplay": "76% (25/33)",
          "bDisplay": "85% (28/33)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Agent 59% to 87%; GPT 5.2 69% to 93%), so this sample cannot separate them.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-same-instance-leaderboard",
          "aN": 33,
          "bN": 33,
          "aContext": "full pipeline on Claude Sonnet 5.5",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.5898,
            0.8717
          ],
          "bRange": [
            0.6908,
            0.9335
          ]
        },
        {
          "metric": "Model calls per instance",
          "aValue": 49.5,
          "bValue": 35.6,
          "unit": "calls",
          "aDisplay": "49.5",
          "bDisplay": "35.6",
          "winner": "unclear",
          "basis": "More or fewer calls is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-model-calls",
          "aN": 33,
          "bN": 33,
          "aContext": "full pipeline on Claude Sonnet 5.5",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances"
        },
        {
          "metric": "Recorded cost per resolved instance: Agent vs the public panel",
          "aValue": 3.706,
          "bValue": 0.628,
          "unit": "usd",
          "aDisplay": "$3.71",
          "bDisplay": "$0.63",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($3.71 vs $0.63, 5.9x) is not tested against run-to-run variation.",
          "studySlug": "cost-thought-experiments",
          "chartId": "cost-per-resolved-agent-vs-panel",
          "aN": 25,
          "bN": 28,
          "aContext": "full pipeline on Claude Sonnet 5.5 · notional",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "calculation": true
        }
      ]
    },
    {
      "slug": "agent-harness-vs-gpt-5-mini",
      "a": "agent-harness",
      "b": "gpt-5-mini",
      "title": "Agent vs GPT 5 mini",
      "seoTitle": "Agent vs GPT 5 mini: measured benchmarks",
      "description": "Agent vs GPT 5 mini: 2 measured metrics from one study (Resolved rate on the same 33 SWE-bench Verified instances; more), with sample sizes and intervals.",
      "verdict": "Agent and GPT 5 mini share 2 measured metrics and 1 list-price calculation from 2 studies. No row separates them: every interval or run range overlaps, too few runs were recorded, no interval was recorded, or more is not better for that metric. The rows are 1 tie and 2 unclear; each row says why. Calculation rows are derived from list prices and recorded counts; they are not bills or runs. Agent is a full pipeline on one model; GPT 5 mini ran under a different, simpler harness, so this compares systems, not models.",
      "rows": [
        {
          "metric": "Resolved rate on the same 33 SWE-bench Verified instances",
          "aValue": 0.7576,
          "bValue": 0.6364,
          "unit": "rate",
          "aDisplay": "76% (25/33)",
          "bDisplay": "64% (21/33)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Agent 59% to 87%; GPT 5 mini 47% to 78%), so this sample cannot separate them.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-same-instance-leaderboard",
          "aN": 33,
          "bN": 33,
          "aContext": "full pipeline on Claude Sonnet 5.5",
          "bContext": "public mini-SWE-agent v2 run, same instances",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.5898,
            0.8717
          ],
          "bRange": [
            0.4662,
            0.7781
          ]
        },
        {
          "metric": "Model calls per instance",
          "aValue": 49.5,
          "bValue": 20.8,
          "unit": "calls",
          "aDisplay": "49.5",
          "bDisplay": "20.8",
          "winner": "unclear",
          "basis": "More or fewer calls is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-model-calls",
          "aN": 33,
          "bN": 33,
          "aContext": "full pipeline on Claude Sonnet 5.5",
          "bContext": "public mini-SWE-agent v2 run, same instances"
        },
        {
          "metric": "Recorded cost per resolved instance: Agent vs the public panel",
          "aValue": 3.706,
          "bValue": 0.08,
          "unit": "usd",
          "aDisplay": "$3.71",
          "bDisplay": "$0.080",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($3.71 vs $0.080, 46x) is not tested against run-to-run variation.",
          "studySlug": "cost-thought-experiments",
          "chartId": "cost-per-resolved-agent-vs-panel",
          "aN": 25,
          "bN": 21,
          "aContext": "full pipeline on Claude Sonnet 5.5 · notional",
          "bContext": "public mini-SWE-agent v2 run, same instances",
          "calculation": true
        }
      ]
    },
    {
      "slug": "agent-harness-vs-kimi-k2-5",
      "a": "agent-harness",
      "b": "kimi-k2-5",
      "title": "Agent vs Kimi K2.5",
      "seoTitle": "Agent vs Kimi K2.5: measured benchmarks",
      "description": "Agent vs Kimi K2.5: 2 measured metrics from one study (Resolved rate on the same 33 SWE-bench Verified instances; more), with sample sizes and intervals.",
      "verdict": "Agent and Kimi K2.5 share 2 measured metrics and 1 list-price calculation from 2 studies. No row separates them: every interval or run range overlaps, too few runs were recorded, no interval was recorded, or more is not better for that metric. The rows are 1 tie and 2 unclear; each row says why. Calculation rows are derived from list prices and recorded counts; they are not bills or runs. Agent is a full pipeline on one model; Kimi K2.5 ran under a different, simpler harness, so this compares systems, not models.",
      "rows": [
        {
          "metric": "Resolved rate on the same 33 SWE-bench Verified instances",
          "aValue": 0.7576,
          "bValue": 0.697,
          "unit": "rate",
          "aDisplay": "76% (25/33)",
          "bDisplay": "70% (23/33)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Agent 59% to 87%; Kimi K2.5 53% to 83%), so this sample cannot separate them.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-same-instance-leaderboard",
          "aN": 33,
          "bN": 33,
          "aContext": "full pipeline on Claude Sonnet 5.5",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.5898,
            0.8717
          ],
          "bRange": [
            0.5266,
            0.8262
          ]
        },
        {
          "metric": "Model calls per instance",
          "aValue": 49.5,
          "bValue": 56.7,
          "unit": "calls",
          "aDisplay": "49.5",
          "bDisplay": "56.7",
          "winner": "unclear",
          "basis": "More or fewer calls is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-model-calls",
          "aN": 33,
          "bN": 33,
          "aContext": "full pipeline on Claude Sonnet 5.5",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances"
        },
        {
          "metric": "Recorded cost per resolved instance: Agent vs the public panel",
          "aValue": 3.706,
          "bValue": 0.256,
          "unit": "usd",
          "aDisplay": "$3.71",
          "bDisplay": "$0.26",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($3.71 vs $0.26, 14x) is not tested against run-to-run variation.",
          "studySlug": "cost-thought-experiments",
          "chartId": "cost-per-resolved-agent-vs-panel",
          "aN": 25,
          "bN": 23,
          "aContext": "full pipeline on Claude Sonnet 5.5 · notional",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "calculation": true
        }
      ]
    },
    {
      "slug": "agent-harness-vs-minimax-m2-5",
      "a": "agent-harness",
      "b": "minimax-m2-5",
      "title": "Agent vs MiniMax M2.5",
      "seoTitle": "Agent vs MiniMax M2.5: measured benchmarks",
      "description": "Agent vs MiniMax M2.5: 2 measured metrics from one study (Resolved rate on the same 33 SWE-bench Verified instances; more), with sample sizes and intervals.",
      "verdict": "Agent and MiniMax M2.5 share 2 measured metrics and 1 list-price calculation from 2 studies. No row separates them: every interval or run range overlaps, too few runs were recorded, no interval was recorded, or more is not better for that metric. The rows are 1 tie and 2 unclear; each row says why. Calculation rows are derived from list prices and recorded counts; they are not bills or runs. Agent is a full pipeline on one model; MiniMax M2.5 ran under a different, simpler harness, so this compares systems, not models.",
      "rows": [
        {
          "metric": "Resolved rate on the same 33 SWE-bench Verified instances",
          "aValue": 0.7576,
          "bValue": 0.697,
          "unit": "rate",
          "aDisplay": "76% (25/33)",
          "bDisplay": "70% (23/33)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Agent 59% to 87%; MiniMax M2.5 53% to 83%), so this sample cannot separate them.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-same-instance-leaderboard",
          "aN": 33,
          "bN": 33,
          "aContext": "full pipeline on Claude Sonnet 5.5",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.5898,
            0.8717
          ],
          "bRange": [
            0.5266,
            0.8262
          ]
        },
        {
          "metric": "Model calls per instance",
          "aValue": 49.5,
          "bValue": 58.4,
          "unit": "calls",
          "aDisplay": "49.5",
          "bDisplay": "58.4",
          "winner": "unclear",
          "basis": "More or fewer calls is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "swe-bench-verified",
          "n": 33,
          "chartId": "swebench-model-calls",
          "aN": 33,
          "bN": 33,
          "aContext": "full pipeline on Claude Sonnet 5.5",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances"
        },
        {
          "metric": "Recorded cost per resolved instance: Agent vs the public panel",
          "aValue": 3.706,
          "bValue": 0.107,
          "unit": "usd",
          "aDisplay": "$3.71",
          "bDisplay": "$0.11",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($3.71 vs $0.11, 35x) is not tested against run-to-run variation.",
          "studySlug": "cost-thought-experiments",
          "chartId": "cost-per-resolved-agent-vs-panel",
          "aN": 25,
          "bN": 23,
          "aContext": "full pipeline on Claude Sonnet 5.5 · notional",
          "bContext": "effort high · public mini-SWE-agent v2 run, same instances",
          "calculation": true
        }
      ]
    },
    {
      "slug": "amazon-bedrock-vs-baseten",
      "a": "amazon-bedrock",
      "b": "baseten",
      "title": "Amazon Bedrock vs Baseten",
      "seoTitle": "Amazon Bedrock vs Baseten: price per model",
      "description": "Amazon Bedrock vs Baseten: 2 list prices per million tokens for the same models, as reported by OpenRouter’s public API, with the gap stated.",
      "verdict": "Amazon Bedrock and Baseten share 2 reported list prices from one study. No row names a winner: a reported list price has no interval, so each gap is stated, not ranked. Prices change often; the snapshot date is in each row’s context. The rows are 2 unclear; each row says why.",
      "rows": [
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Input)",
          "aValue": 0.15,
          "bValue": 0.1,
          "unit": "usd",
          "aDisplay": "$0.15",
          "bDisplay": "$0.10",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.15 vs $0.10, 1.5x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Output)",
          "aValue": 0.6,
          "bValue": 0.5,
          "unit": "usd",
          "aDisplay": "$0.60",
          "bDisplay": "$0.50",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.60 vs $0.50, 1.2x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "amazon-bedrock-vs-cerebras",
      "a": "amazon-bedrock",
      "b": "cerebras",
      "title": "Amazon Bedrock vs Cerebras",
      "seoTitle": "Amazon Bedrock vs Cerebras: price per model",
      "description": "Amazon Bedrock vs Cerebras: 2 list prices per million tokens for the same models, as reported by OpenRouter’s public API, with the gap stated.",
      "verdict": "Amazon Bedrock and Cerebras share 2 reported list prices from one study. No row names a winner: a reported list price has no interval, so each gap is stated, not ranked. Prices change often; the snapshot date is in each row’s context. The rows are 2 unclear; each row says why.",
      "rows": [
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Input)",
          "aValue": 0.15,
          "bValue": 0.35,
          "unit": "usd",
          "aDisplay": "$0.15",
          "bDisplay": "$0.35",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.15 vs $0.35, 2.3x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp16 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Output)",
          "aValue": 0.6,
          "bValue": 0.75,
          "unit": "usd",
          "aDisplay": "$0.60",
          "bDisplay": "$0.75",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.60 vs $0.75, 1.3x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp16 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "amazon-bedrock-vs-deepinfra",
      "a": "amazon-bedrock",
      "b": "deepinfra",
      "title": "Amazon Bedrock vs DeepInfra",
      "seoTitle": "Amazon Bedrock vs DeepInfra: price per model",
      "description": "Amazon Bedrock vs DeepInfra: 2 list prices per million tokens for the same models, as reported by OpenRouter’s public API, with the gap stated.",
      "verdict": "Amazon Bedrock and DeepInfra share 2 reported list prices from one study. No row names a winner: a reported list price has no interval, so each gap is stated, not ranked. Prices change often; the snapshot date is in each row’s context. The rows are 2 unclear; each row says why.",
      "rows": [
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Input)",
          "aValue": 0.15,
          "bValue": 0.037,
          "unit": "usd",
          "aDisplay": "$0.15",
          "bDisplay": "$0.037",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.15 vs $0.037, 4.1x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "bf16 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Output)",
          "aValue": 0.6,
          "bValue": 0.17,
          "unit": "usd",
          "aDisplay": "$0.60",
          "bDisplay": "$0.17",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.60 vs $0.17, 3.5x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "bf16 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "amazon-bedrock-vs-groq",
      "a": "amazon-bedrock",
      "b": "groq",
      "title": "Amazon Bedrock vs Groq",
      "seoTitle": "Amazon Bedrock vs Groq: price per model",
      "description": "Amazon Bedrock vs Groq: 2 list prices per million tokens for the same models, as reported by OpenRouter’s public API, with the gap stated.",
      "verdict": "Amazon Bedrock and Groq share 2 reported list prices from one study. No row names a winner: a reported list price has no interval, so each gap is stated, not ranked. Prices change often; the snapshot date is in each row’s context. The rows are 2 ties; each row says why.",
      "rows": [
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Input)",
          "aValue": 0.15,
          "bValue": 0.15,
          "unit": "usd",
          "aDisplay": "$0.15",
          "bDisplay": "$0.15",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Output)",
          "aValue": 0.6,
          "bValue": 0.6,
          "unit": "usd",
          "aDisplay": "$0.60",
          "bDisplay": "$0.60",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "amazon-bedrock-vs-nebius",
      "a": "amazon-bedrock",
      "b": "nebius",
      "title": "Amazon Bedrock vs Nebius",
      "seoTitle": "Amazon Bedrock vs Nebius: price per model",
      "description": "Amazon Bedrock vs Nebius: 2 list prices per million tokens for the same models, as reported by OpenRouter’s public API, with the gap stated.",
      "verdict": "Amazon Bedrock and Nebius share 2 reported list prices from one study. No row names a winner: a reported list price has no interval, so each gap is stated, not ranked. Prices change often; the snapshot date is in each row’s context. The rows are 2 ties; each row says why.",
      "rows": [
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Input)",
          "aValue": 0.15,
          "bValue": 0.15,
          "unit": "usd",
          "aDisplay": "$0.15",
          "bDisplay": "$0.15",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Output)",
          "aValue": 0.6,
          "bValue": 0.6,
          "unit": "usd",
          "aDisplay": "$0.60",
          "bDisplay": "$0.60",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "amazon-bedrock-vs-novita",
      "a": "amazon-bedrock",
      "b": "novita",
      "title": "Amazon Bedrock vs Novita AI",
      "seoTitle": "Amazon Bedrock vs Novita AI: price per model",
      "description": "Amazon Bedrock vs Novita AI: 2 list prices per million tokens for the same models, as reported by OpenRouter’s public API, with the gap stated.",
      "verdict": "Amazon Bedrock and Novita AI share 2 reported list prices from one study. No row names a winner: a reported list price has no interval, so each gap is stated, not ranked. Prices change often; the snapshot date is in each row’s context. The rows are 2 unclear; each row says why.",
      "rows": [
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Input)",
          "aValue": 0.15,
          "bValue": 0.05,
          "unit": "usd",
          "aDisplay": "$0.15",
          "bDisplay": "$0.050",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.15 vs $0.050, 3.0x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Output)",
          "aValue": 0.6,
          "bValue": 0.25,
          "unit": "usd",
          "aDisplay": "$0.60",
          "bDisplay": "$0.25",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.60 vs $0.25, 2.4x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "amazon-bedrock-vs-parasail",
      "a": "amazon-bedrock",
      "b": "parasail",
      "title": "Amazon Bedrock vs Parasail",
      "seoTitle": "Amazon Bedrock vs Parasail: price per model",
      "description": "Amazon Bedrock vs Parasail: 2 list prices per million tokens for the same models, as reported by OpenRouter’s public API, with the gap stated.",
      "verdict": "Amazon Bedrock and Parasail share 2 reported list prices from one study. No row names a winner: a reported list price has no interval, so each gap is stated, not ranked. Prices change often; the snapshot date is in each row’s context. The rows are 2 unclear; each row says why.",
      "rows": [
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Input)",
          "aValue": 0.15,
          "bValue": 0.1,
          "unit": "usd",
          "aDisplay": "$0.15",
          "bDisplay": "$0.10",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.15 vs $0.10, 1.5x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Output)",
          "aValue": 0.6,
          "bValue": 0.75,
          "unit": "usd",
          "aDisplay": "$0.60",
          "bDisplay": "$0.75",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.60 vs $0.75, 1.3x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "amazon-bedrock-vs-sambanova",
      "a": "amazon-bedrock",
      "b": "sambanova",
      "title": "Amazon Bedrock vs SambaNova",
      "seoTitle": "Amazon Bedrock vs SambaNova: price per model",
      "description": "Amazon Bedrock vs SambaNova: 2 list prices per million tokens for the same models, as reported by OpenRouter’s public API, with the gap stated.",
      "verdict": "Amazon Bedrock and SambaNova share 2 reported list prices from one study. No row names a winner: a reported list price has no interval, so each gap is stated, not ranked. Prices change often; the snapshot date is in each row’s context. The rows are 2 unclear; each row says why.",
      "rows": [
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Input)",
          "aValue": 0.15,
          "bValue": 0.14,
          "unit": "usd",
          "aDisplay": "$0.15",
          "bDisplay": "$0.14",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.15 vs $0.14, 1.1x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Output)",
          "aValue": 0.6,
          "bValue": 0.95,
          "unit": "usd",
          "aDisplay": "$0.60",
          "bDisplay": "$0.95",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.60 vs $0.95, 1.6x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "amazon-bedrock-vs-siliconflow",
      "a": "amazon-bedrock",
      "b": "siliconflow",
      "title": "Amazon Bedrock vs SiliconFlow",
      "seoTitle": "Amazon Bedrock vs SiliconFlow: price per model",
      "description": "Amazon Bedrock vs SiliconFlow: 2 list prices per million tokens for the same models, as reported by OpenRouter’s public API, with the gap stated.",
      "verdict": "Amazon Bedrock and SiliconFlow share 2 reported list prices from one study. No row names a winner: a reported list price has no interval, so each gap is stated, not ranked. Prices change often; the snapshot date is in each row’s context. The rows are 2 ties; each row says why.",
      "rows": [
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Input)",
          "aValue": 0.15,
          "bValue": 0.15,
          "unit": "usd",
          "aDisplay": "$0.15",
          "bDisplay": "$0.15",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Output)",
          "aValue": 0.6,
          "bValue": 0.6,
          "unit": "usd",
          "aDisplay": "$0.60",
          "bDisplay": "$0.60",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "amazon-bedrock-vs-together",
      "a": "amazon-bedrock",
      "b": "together",
      "title": "Amazon Bedrock vs Together AI",
      "seoTitle": "Amazon Bedrock vs Together AI: price per model",
      "description": "Amazon Bedrock vs Together AI: 2 list prices per million tokens for the same models, as reported by OpenRouter’s public API, with the gap stated.",
      "verdict": "Amazon Bedrock and Together AI share 2 reported list prices from one study. No row names a winner: a reported list price has no interval, so each gap is stated, not ranked. Prices change often; the snapshot date is in each row’s context. The rows are 2 ties; each row says why.",
      "rows": [
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Input)",
          "aValue": 0.15,
          "bValue": 0.15,
          "unit": "usd",
          "aDisplay": "$0.15",
          "bDisplay": "$0.15",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Output)",
          "aValue": 0.6,
          "bValue": 0.6,
          "unit": "usd",
          "aDisplay": "$0.60",
          "bDisplay": "$0.60",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "cerebras-vs-nebius",
      "a": "cerebras",
      "b": "nebius",
      "title": "Cerebras vs Nebius",
      "seoTitle": "Cerebras vs Nebius: price per model",
      "description": "Cerebras vs Nebius: 2 list prices per million tokens for the same models, as reported by OpenRouter’s public API, with the gap stated.",
      "verdict": "Cerebras and Nebius share 2 reported list prices from one study. No row names a winner: a reported list price has no interval, so each gap is stated, not ranked. Prices change often; the snapshot date is in each row’s context. The rows are 2 unclear; each row says why.",
      "rows": [
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Input)",
          "aValue": 0.35,
          "bValue": 0.15,
          "unit": "usd",
          "aDisplay": "$0.35",
          "bDisplay": "$0.15",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.35 vs $0.15, 2.3x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "fp16 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Output)",
          "aValue": 0.75,
          "bValue": 0.6,
          "unit": "usd",
          "aDisplay": "$0.75",
          "bDisplay": "$0.60",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.75 vs $0.60, 1.3x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "fp16 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "cerebras-vs-novita",
      "a": "cerebras",
      "b": "novita",
      "title": "Cerebras vs Novita AI",
      "seoTitle": "Cerebras vs Novita AI: price per model",
      "description": "Cerebras vs Novita AI: 2 list prices per million tokens for the same models, as reported by OpenRouter’s public API, with the gap stated.",
      "verdict": "Cerebras and Novita AI share 2 reported list prices from one study. No row names a winner: a reported list price has no interval, so each gap is stated, not ranked. Prices change often; the snapshot date is in each row’s context. The rows are 2 unclear; each row says why.",
      "rows": [
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Input)",
          "aValue": 0.35,
          "bValue": 0.05,
          "unit": "usd",
          "aDisplay": "$0.35",
          "bDisplay": "$0.050",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.35 vs $0.050, 7.0x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "fp16 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Output)",
          "aValue": 0.75,
          "bValue": 0.25,
          "unit": "usd",
          "aDisplay": "$0.75",
          "bDisplay": "$0.25",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.75 vs $0.25, 3.0x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "fp16 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "cerebras-vs-sambanova",
      "a": "cerebras",
      "b": "sambanova",
      "title": "Cerebras vs SambaNova",
      "seoTitle": "Cerebras vs SambaNova: price per model",
      "description": "Cerebras vs SambaNova: 2 list prices per million tokens for the same models, as reported by OpenRouter’s public API, with the gap stated.",
      "verdict": "Cerebras and SambaNova share 2 reported list prices from one study. No row names a winner: a reported list price has no interval, so each gap is stated, not ranked. Prices change often; the snapshot date is in each row’s context. The rows are 2 unclear; each row says why.",
      "rows": [
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Input)",
          "aValue": 0.35,
          "bValue": 0.14,
          "unit": "usd",
          "aDisplay": "$0.35",
          "bDisplay": "$0.14",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.35 vs $0.14, 2.5x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "fp16 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Output)",
          "aValue": 0.75,
          "bValue": 0.95,
          "unit": "usd",
          "aDisplay": "$0.75",
          "bDisplay": "$0.95",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.75 vs $0.95, 1.3x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "fp16 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "deepinfra-vs-cerebras",
      "a": "deepinfra",
      "b": "cerebras",
      "title": "DeepInfra vs Cerebras",
      "seoTitle": "DeepInfra vs Cerebras: price per model",
      "description": "DeepInfra vs Cerebras: 2 list prices per million tokens for the same models, as reported by OpenRouter’s public API, with the gap stated.",
      "verdict": "DeepInfra and Cerebras share 2 reported list prices from one study. No row names a winner: a reported list price has no interval, so each gap is stated, not ranked. Prices change often; the snapshot date is in each row’s context. The rows are 2 unclear; each row says why.",
      "rows": [
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Input)",
          "aValue": 0.037,
          "bValue": 0.35,
          "unit": "usd",
          "aDisplay": "$0.037",
          "bDisplay": "$0.35",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.037 vs $0.35, 9.5x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "bf16 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp16 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Output)",
          "aValue": 0.17,
          "bValue": 0.75,
          "unit": "usd",
          "aDisplay": "$0.17",
          "bDisplay": "$0.75",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.17 vs $0.75, 4.4x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "bf16 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp16 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "fireworks-vs-nebius",
      "a": "fireworks",
      "b": "nebius",
      "title": "Fireworks AI vs Nebius",
      "seoTitle": "Fireworks AI vs Nebius: price per model",
      "description": "Fireworks AI vs Nebius: 2 list prices per million tokens for the same models, as reported by OpenRouter’s public API, with the gap stated.",
      "verdict": "Fireworks AI and Nebius share 2 reported list prices from one study. No row names a winner: a reported list price has no interval, so each gap is stated, not ranked. Prices change often; the snapshot date is in each row’s context. The rows are 2 ties; each row says why.",
      "rows": [
        {
          "metric": "GLM 5.3: price per million tokens by provider (Input)",
          "aValue": 1.4,
          "bValue": 1.4,
          "unit": "usd",
          "aDisplay": "$1.40",
          "bDisplay": "$1.40",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "aContext": "GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp4 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "GLM 5.3: price per million tokens by provider (Output)",
          "aValue": 4.4,
          "bValue": 4.4,
          "unit": "usd",
          "aDisplay": "$4.40",
          "bDisplay": "$4.40",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "aContext": "GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp4 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "google-vertex-vs-baseten",
      "a": "google-vertex",
      "b": "baseten",
      "title": "Google Vertex AI vs Baseten",
      "seoTitle": "Google Vertex AI vs Baseten: price per model",
      "description": "Google Vertex AI vs Baseten: 2 list prices per million tokens for the same models, as reported by OpenRouter’s public API, with the gap stated.",
      "verdict": "Google Vertex AI and Baseten share 2 reported list prices from one study. No row names a winner: a reported list price has no interval, so each gap is stated, not ranked. Prices change often; the snapshot date is in each row’s context. The rows are 2 unclear; each row says why.",
      "rows": [
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Input)",
          "aValue": 0.09,
          "bValue": 0.1,
          "unit": "usd",
          "aDisplay": "$0.090",
          "bDisplay": "$0.10",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.090 vs $0.10, 1.1x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Output)",
          "aValue": 0.36,
          "bValue": 0.5,
          "unit": "usd",
          "aDisplay": "$0.36",
          "bDisplay": "$0.50",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.36 vs $0.50, 1.4x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "google-vertex-vs-cerebras",
      "a": "google-vertex",
      "b": "cerebras",
      "title": "Google Vertex AI vs Cerebras",
      "seoTitle": "Google Vertex AI vs Cerebras: price per model",
      "description": "Google Vertex AI vs Cerebras: 2 list prices per million tokens for the same models, as reported by OpenRouter’s public API, with the gap stated.",
      "verdict": "Google Vertex AI and Cerebras share 2 reported list prices from one study. No row names a winner: a reported list price has no interval, so each gap is stated, not ranked. Prices change often; the snapshot date is in each row’s context. The rows are 2 unclear; each row says why.",
      "rows": [
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Input)",
          "aValue": 0.09,
          "bValue": 0.35,
          "unit": "usd",
          "aDisplay": "$0.090",
          "bDisplay": "$0.35",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.090 vs $0.35, 3.9x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp16 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Output)",
          "aValue": 0.36,
          "bValue": 0.75,
          "unit": "usd",
          "aDisplay": "$0.36",
          "bDisplay": "$0.75",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.36 vs $0.75, 2.1x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp16 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "google-vertex-vs-cloudflare",
      "a": "google-vertex",
      "b": "cloudflare",
      "title": "Google Vertex AI vs Cloudflare Workers AI",
      "seoTitle": "Google Vertex AI vs Cloudflare Workers AI: price per model",
      "description": "Google Vertex AI vs Cloudflare Workers AI: 2 list prices per million tokens for the same models, as reported by OpenRouter’s public API, with the gap stated.",
      "verdict": "Google Vertex AI and Cloudflare Workers AI share 2 reported list prices from one study. No row names a winner: a reported list price has no interval, so each gap is stated, not ranked. Prices change often; the snapshot date is in each row’s context. The rows are 2 unclear; each row says why.",
      "rows": [
        {
          "metric": "Llama 3.3 70B Instruct: price per million tokens by provider (Input)",
          "aValue": 0.72,
          "bValue": 0.293,
          "unit": "usd",
          "aDisplay": "$0.72",
          "bDisplay": "$0.29",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.72 vs $0.29, 2.5x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-llama-3-3-70b-instruct",
          "aContext": "Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Llama 3.3 70B Instruct: price per million tokens by provider (Output)",
          "aValue": 0.72,
          "bValue": 2.253,
          "unit": "usd",
          "aDisplay": "$0.72",
          "bDisplay": "$2.25",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.72 vs $2.25, 3.1x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-llama-3-3-70b-instruct",
          "aContext": "Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "google-vertex-vs-nebius",
      "a": "google-vertex",
      "b": "nebius",
      "title": "Google Vertex AI vs Nebius",
      "seoTitle": "Google Vertex AI vs Nebius: price per model",
      "description": "Google Vertex AI vs Nebius: 2 list prices per million tokens for the same models, as reported by OpenRouter’s public API, with the gap stated.",
      "verdict": "Google Vertex AI and Nebius share 2 reported list prices from one study. No row names a winner: a reported list price has no interval, so each gap is stated, not ranked. Prices change often; the snapshot date is in each row’s context. The rows are 2 unclear; each row says why.",
      "rows": [
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Input)",
          "aValue": 0.09,
          "bValue": 0.15,
          "unit": "usd",
          "aDisplay": "$0.090",
          "bDisplay": "$0.15",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.090 vs $0.15, 1.7x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Output)",
          "aValue": 0.36,
          "bValue": 0.6,
          "unit": "usd",
          "aDisplay": "$0.36",
          "bDisplay": "$0.60",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.36 vs $0.60, 1.7x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "google-vertex-vs-siliconflow",
      "a": "google-vertex",
      "b": "siliconflow",
      "title": "Google Vertex AI vs SiliconFlow",
      "seoTitle": "Google Vertex AI vs SiliconFlow: price per model",
      "description": "Google Vertex AI vs SiliconFlow: 2 list prices per million tokens for the same models, as reported by OpenRouter’s public API, with the gap stated.",
      "verdict": "Google Vertex AI and SiliconFlow share 2 reported list prices from one study. No row names a winner: a reported list price has no interval, so each gap is stated, not ranked. Prices change often; the snapshot date is in each row’s context. The rows are 2 unclear; each row says why.",
      "rows": [
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Input)",
          "aValue": 0.09,
          "bValue": 0.15,
          "unit": "usd",
          "aDisplay": "$0.090",
          "bDisplay": "$0.15",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.090 vs $0.15, 1.7x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Output)",
          "aValue": 0.36,
          "bValue": 0.6,
          "unit": "usd",
          "aDisplay": "$0.36",
          "bDisplay": "$0.60",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.36 vs $0.60, 1.7x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "groq-vs-cloudflare",
      "a": "groq",
      "b": "cloudflare",
      "title": "Groq vs Cloudflare Workers AI",
      "seoTitle": "Groq vs Cloudflare Workers AI: price per model",
      "description": "Groq vs Cloudflare Workers AI: 2 list prices per million tokens for the same models, as reported by OpenRouter’s public API, with the gap stated.",
      "verdict": "Groq and Cloudflare Workers AI share 2 reported list prices from one study. No row names a winner: a reported list price has no interval, so each gap is stated, not ranked. Prices change often; the snapshot date is in each row’s context. The rows are 2 unclear; each row says why.",
      "rows": [
        {
          "metric": "Llama 3.3 70B Instruct: price per million tokens by provider (Input)",
          "aValue": 0.59,
          "bValue": 0.293,
          "unit": "usd",
          "aDisplay": "$0.59",
          "bDisplay": "$0.29",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.59 vs $0.29, 2.0x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-llama-3-3-70b-instruct",
          "aContext": "Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Llama 3.3 70B Instruct: price per million tokens by provider (Output)",
          "aValue": 0.79,
          "bValue": 2.253,
          "unit": "usd",
          "aDisplay": "$0.79",
          "bDisplay": "$2.25",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.79 vs $2.25, 2.9x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-llama-3-3-70b-instruct",
          "aContext": "Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "groq-vs-nebius",
      "a": "groq",
      "b": "nebius",
      "title": "Groq vs Nebius",
      "seoTitle": "Groq vs Nebius: price per model",
      "description": "Groq vs Nebius: 2 list prices per million tokens for the same models, as reported by OpenRouter’s public API, with the gap stated.",
      "verdict": "Groq and Nebius share 2 reported list prices from one study. No row names a winner: a reported list price has no interval, so each gap is stated, not ranked. Prices change often; the snapshot date is in each row’s context. The rows are 2 ties; each row says why.",
      "rows": [
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Input)",
          "aValue": 0.15,
          "bValue": 0.15,
          "unit": "usd",
          "aDisplay": "$0.15",
          "bDisplay": "$0.15",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Output)",
          "aValue": 0.6,
          "bValue": 0.6,
          "unit": "usd",
          "aDisplay": "$0.60",
          "bDisplay": "$0.60",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "nebius-vs-cloudflare",
      "a": "nebius",
      "b": "cloudflare",
      "title": "Nebius vs Cloudflare Workers AI",
      "seoTitle": "Nebius vs Cloudflare Workers AI: price per model",
      "description": "Nebius vs Cloudflare Workers AI: 2 list prices per million tokens for the same models, as reported by OpenRouter’s public API, with the gap stated.",
      "verdict": "Nebius and Cloudflare Workers AI share 2 reported list prices from one study. No row names a winner: a reported list price has no interval, so each gap is stated, not ranked. Prices change often; the snapshot date is in each row’s context. The rows are 2 ties; each row says why.",
      "rows": [
        {
          "metric": "GLM 5.3: price per million tokens by provider (Input)",
          "aValue": 1.4,
          "bValue": 1.4,
          "unit": "usd",
          "aDisplay": "$1.40",
          "bDisplay": "$1.40",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "aContext": "fp4 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "GLM 5.3: price per million tokens by provider (Output)",
          "aValue": 4.4,
          "bValue": 4.4,
          "unit": "usd",
          "aDisplay": "$4.40",
          "bDisplay": "$4.40",
          "winner": "tie",
          "basis": "Same reported list price.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-glm-5-3",
          "aContext": "fp4 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "sambanova-vs-baseten",
      "a": "sambanova",
      "b": "baseten",
      "title": "SambaNova vs Baseten",
      "seoTitle": "SambaNova vs Baseten: price per model",
      "description": "SambaNova vs Baseten: 2 list prices per million tokens for the same models, as reported by OpenRouter’s public API, with the gap stated.",
      "verdict": "SambaNova and Baseten share 2 reported list prices from one study. No row names a winner: a reported list price has no interval, so each gap is stated, not ranked. Prices change often; the snapshot date is in each row’s context. The rows are 2 unclear; each row says why.",
      "rows": [
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Input)",
          "aValue": 0.14,
          "bValue": 0.1,
          "unit": "usd",
          "aDisplay": "$0.14",
          "bDisplay": "$0.10",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.14 vs $0.10, 1.4x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Output)",
          "aValue": 0.95,
          "bValue": 0.5,
          "unit": "usd",
          "aDisplay": "$0.95",
          "bDisplay": "$0.50",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.95 vs $0.50, 1.9x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "sambanova-vs-cloudflare",
      "a": "sambanova",
      "b": "cloudflare",
      "title": "SambaNova vs Cloudflare Workers AI",
      "seoTitle": "SambaNova vs Cloudflare Workers AI: price per model",
      "description": "SambaNova vs Cloudflare Workers AI: 2 list prices per million tokens for the same models, as reported by OpenRouter’s public API, with the gap stated.",
      "verdict": "SambaNova and Cloudflare Workers AI share 2 reported list prices from one study. No row names a winner: a reported list price has no interval, so each gap is stated, not ranked. Prices change often; the snapshot date is in each row’s context. The rows are 2 unclear; each row says why.",
      "rows": [
        {
          "metric": "Llama 3.3 70B Instruct: price per million tokens by provider (Input)",
          "aValue": 0.45,
          "bValue": 0.293,
          "unit": "usd",
          "aDisplay": "$0.45",
          "bDisplay": "$0.29",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.45 vs $0.29, 1.5x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-llama-3-3-70b-instruct",
          "aContext": "Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "Llama 3.3 70B Instruct: price per million tokens by provider (Output)",
          "aValue": 0.9,
          "bValue": 2.253,
          "unit": "usd",
          "aDisplay": "$0.90",
          "bDisplay": "$2.25",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.90 vs $2.25, 2.5x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-llama-3-3-70b-instruct",
          "aContext": "Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "sambanova-vs-nebius",
      "a": "sambanova",
      "b": "nebius",
      "title": "SambaNova vs Nebius",
      "seoTitle": "SambaNova vs Nebius: price per model",
      "description": "SambaNova vs Nebius: 2 list prices per million tokens for the same models, as reported by OpenRouter’s public API, with the gap stated.",
      "verdict": "SambaNova and Nebius share 2 reported list prices from one study. No row names a winner: a reported list price has no interval, so each gap is stated, not ranked. Prices change often; the snapshot date is in each row’s context. The rows are 2 unclear; each row says why.",
      "rows": [
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Input)",
          "aValue": 0.14,
          "bValue": 0.15,
          "unit": "usd",
          "aDisplay": "$0.14",
          "bDisplay": "$0.15",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.14 vs $0.15, 1.1x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Output)",
          "aValue": 0.95,
          "bValue": 0.6,
          "unit": "usd",
          "aDisplay": "$0.95",
          "bDisplay": "$0.60",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.95 vs $0.60, 1.6x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "sambanova-vs-siliconflow",
      "a": "sambanova",
      "b": "siliconflow",
      "title": "SambaNova vs SiliconFlow",
      "seoTitle": "SambaNova vs SiliconFlow: price per model",
      "description": "SambaNova vs SiliconFlow: 2 list prices per million tokens for the same models, as reported by OpenRouter’s public API, with the gap stated.",
      "verdict": "SambaNova and SiliconFlow share 2 reported list prices from one study. No row names a winner: a reported list price has no interval, so each gap is stated, not ranked. Prices change often; the snapshot date is in each row’s context. The rows are 2 unclear; each row says why.",
      "rows": [
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Input)",
          "aValue": 0.14,
          "bValue": 0.15,
          "unit": "usd",
          "aDisplay": "$0.14",
          "bDisplay": "$0.15",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.14 vs $0.15, 1.1x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Output)",
          "aValue": 0.95,
          "bValue": 0.6,
          "unit": "usd",
          "aDisplay": "$0.95",
          "bDisplay": "$0.60",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.95 vs $0.60, 1.6x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp8 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    },
    {
      "slug": "together-vs-cerebras",
      "a": "together",
      "b": "cerebras",
      "title": "Together AI vs Cerebras",
      "seoTitle": "Together AI vs Cerebras: price per model",
      "description": "Together AI vs Cerebras: 2 list prices per million tokens for the same models, as reported by OpenRouter’s public API, with the gap stated.",
      "verdict": "Together AI and Cerebras share 2 reported list prices from one study. No row names a winner: a reported list price has no interval, so each gap is stated, not ranked. Prices change often; the snapshot date is in each row’s context. The rows are 2 unclear; each row says why.",
      "rows": [
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Input)",
          "aValue": 0.15,
          "bValue": 0.35,
          "unit": "usd",
          "aDisplay": "$0.15",
          "bDisplay": "$0.35",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.15 vs $0.35, 2.3x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp16 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        },
        {
          "metric": "gpt-oss-120b: price per million tokens by provider (Output)",
          "aValue": 0.6,
          "bValue": 0.75,
          "unit": "usd",
          "aDisplay": "$0.60",
          "bDisplay": "$0.75",
          "winner": "unclear",
          "basis": "A reported list price has no interval, so the gap ($0.60 vs $0.75, 1.3x) is stated, not ranked. Check quantization and context before treating the two as equal products.",
          "studySlug": "inference-provider-index",
          "chartId": "provider-prices-gpt-oss-120b",
          "aContext": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06",
          "bContext": "fp16 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
        }
      ]
    }
  ],
  "effortComparisons": [
    {
      "slug": "claude-sonnet-5-5-low-vs-medium",
      "entity": "claude-sonnet-5-5",
      "aEffort": "low",
      "bEffort": "medium",
      "title": "Claude Sonnet 5.5: low vs medium effort",
      "seoTitle": "Claude Sonnet 5.5: low vs medium effort, measured",
      "description": "Claude Sonnet 5.5 at low vs medium effort: 3 measured metrics from one study, with sample sizes, intervals and every failure counted.",
      "verdict": "Claude Sonnet 5.5 at low effort and Claude Sonnet 5.5 at medium effort share 3 measured metrics and 3 list-price calculations from 2 studies. Only the effort setting differs between the two sides of a row; the route and the task set are the same. No row separates them: every interval or run range overlaps, too few runs were recorded, no interval was recorded, or more is not better for that metric. The rows are 1 tie and 5 unclear; each row says why. Calculation rows are derived from list prices and recorded counts; they are not bills or runs.",
      "rows": [
        {
          "metric": "Strict pass rate by effort on eight hard tasks",
          "aValue": 1,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "100% (16/16)",
          "bDisplay": "100% (16/16)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Sonnet 5.5 at low effort 81% to 100%; Claude Sonnet 5.5 at medium effort 81% to 100%), so this sample cannot separate them.",
          "studySlug": "effort-ladder",
          "n": 16,
          "chartId": "effort-ladder-pass-rate",
          "aN": 16,
          "bN": 16,
          "aContext": "Claude Code · effort low · eight hard validated tasks, effort ladder",
          "bContext": "Claude Code · effort medium · eight hard validated tasks, effort ladder",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.8064,
            1
          ],
          "bRange": [
            0.8064,
            1
          ]
        },
        {
          "metric": "Total time per call by effort on hard tasks",
          "aValue": 5.82,
          "bValue": 7.63,
          "unit": "seconds",
          "aDisplay": "5.82 s",
          "bDisplay": "7.63 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Sonnet 5.5 at low effort 2.78 s to 20.0 s; Claude Sonnet 5.5 at medium effort 2.71 s to 24.0 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "effort-ladder",
          "n": 16,
          "chartId": "effort-ladder-total-latency",
          "aN": 16,
          "bN": 16,
          "aContext": "Claude Code · effort low · eight hard validated tasks, effort ladder",
          "bContext": "Claude Code · effort medium · eight hard validated tasks, effort ladder",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            2.78,
            19.96
          ],
          "bRange": [
            2.71,
            24.01
          ]
        },
        {
          "metric": "Output tokens per call by effort on hard tasks (Output tokens)",
          "aValue": 667,
          "bValue": 770,
          "unit": "tokens",
          "aDisplay": "667",
          "bDisplay": "770",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "effort-ladder",
          "n": 16,
          "chartId": "effort-ladder-output-tokens",
          "aN": 16,
          "bN": 16,
          "aContext": "Claude Code · effort low · eight hard validated tasks, effort ladder",
          "bContext": "Claude Code · effort medium · eight hard validated tasks, effort ladder"
        },
        {
          "metric": "List-price cost per strict pass by effort (calculation)",
          "aValue": 0.01219,
          "bValue": 0.01352,
          "unit": "usd",
          "aDisplay": "$0.012",
          "bDisplay": "$0.014",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.012 vs $0.014) is not tested against run-to-run variation.",
          "studySlug": "effort-ladder",
          "n": 16,
          "chartId": "effort-ladder-cost-per-pass",
          "aN": 16,
          "bN": 16,
          "aContext": "Claude Code · effort low · eight hard validated tasks, effort ladder",
          "bContext": "Claude Code · effort medium · eight hard validated tasks, effort ladder",
          "calculation": true
        },
        {
          "metric": "Reasoning cost per strict pass by effort, with the total (calculation) (Reasoning cost per strict pass)",
          "aValue": 0.004317,
          "bValue": 0.005946,
          "unit": "usd",
          "aDisplay": "$0.0043",
          "bDisplay": "$0.0059",
          "winner": "unclear",
          "basis": "More or fewer usd is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "n": 16,
          "chartId": "thinking-bill-by-effort",
          "aN": 16,
          "bN": 16,
          "aContext": "Claude Code · effort low",
          "bContext": "Claude Code · effort medium",
          "calculation": true
        },
        {
          "metric": "Reasoning cost per strict pass by effort, with the total (calculation) (Total cost per strict pass)",
          "aValue": 0.012191,
          "bValue": 0.01352,
          "unit": "usd",
          "aDisplay": "$0.012",
          "bDisplay": "$0.014",
          "winner": "unclear",
          "basis": "More or fewer usd is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "n": 16,
          "chartId": "thinking-bill-by-effort",
          "aN": 16,
          "bN": 16,
          "aContext": "Claude Code · effort low",
          "bContext": "Claude Code · effort medium",
          "calculation": true
        }
      ]
    },
    {
      "slug": "claude-sonnet-5-5-low-vs-high",
      "entity": "claude-sonnet-5-5",
      "aEffort": "low",
      "bEffort": "high",
      "title": "Claude Sonnet 5.5: low vs high effort",
      "seoTitle": "Claude Sonnet 5.5: low vs high effort, measured",
      "description": "Claude Sonnet 5.5 at low vs high effort: 3 measured metrics from one study, with sample sizes, intervals and every failure counted.",
      "verdict": "Claude Sonnet 5.5 at low effort and Claude Sonnet 5.5 at high effort share 3 measured metrics and 3 list-price calculations from 2 studies. Only the effort setting differs between the two sides of a row; the route and the task set are the same. No row separates them: every interval or run range overlaps, too few runs were recorded, no interval was recorded, or more is not better for that metric. The rows are 1 tie and 5 unclear; each row says why. Calculation rows are derived from list prices and recorded counts; they are not bills or runs.",
      "rows": [
        {
          "metric": "Strict pass rate by effort on eight hard tasks",
          "aValue": 1,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "100% (16/16)",
          "bDisplay": "100% (16/16)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Sonnet 5.5 at low effort 81% to 100%; Claude Sonnet 5.5 at high effort 81% to 100%), so this sample cannot separate them.",
          "studySlug": "effort-ladder",
          "n": 16,
          "chartId": "effort-ladder-pass-rate",
          "aN": 16,
          "bN": 16,
          "aContext": "Claude Code · effort low · eight hard validated tasks, effort ladder",
          "bContext": "Claude Code · effort high · eight hard validated tasks, effort ladder",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.8064,
            1
          ],
          "bRange": [
            0.8064,
            1
          ]
        },
        {
          "metric": "Total time per call by effort on hard tasks",
          "aValue": 5.82,
          "bValue": 8.81,
          "unit": "seconds",
          "aDisplay": "5.82 s",
          "bDisplay": "8.81 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Sonnet 5.5 at low effort 2.78 s to 20.0 s; Claude Sonnet 5.5 at high effort 2.93 s to 35.8 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "effort-ladder",
          "n": 16,
          "chartId": "effort-ladder-total-latency",
          "aN": 16,
          "bN": 16,
          "aContext": "Claude Code · effort low · eight hard validated tasks, effort ladder",
          "bContext": "Claude Code · effort high · eight hard validated tasks, effort ladder",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            2.78,
            19.96
          ],
          "bRange": [
            2.93,
            35.81
          ]
        },
        {
          "metric": "Output tokens per call by effort on hard tasks (Output tokens)",
          "aValue": 667,
          "bValue": 1192,
          "unit": "tokens",
          "aDisplay": "667",
          "bDisplay": "1,192",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "effort-ladder",
          "n": 16,
          "chartId": "effort-ladder-output-tokens",
          "aN": 16,
          "bN": 16,
          "aContext": "Claude Code · effort low · eight hard validated tasks, effort ladder",
          "bContext": "Claude Code · effort high · eight hard validated tasks, effort ladder"
        },
        {
          "metric": "List-price cost per strict pass by effort (calculation)",
          "aValue": 0.01219,
          "bValue": 0.01671,
          "unit": "usd",
          "aDisplay": "$0.012",
          "bDisplay": "$0.017",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.012 vs $0.017) is not tested against run-to-run variation.",
          "studySlug": "effort-ladder",
          "n": 16,
          "chartId": "effort-ladder-cost-per-pass",
          "aN": 16,
          "bN": 16,
          "aContext": "Claude Code · effort low · eight hard validated tasks, effort ladder",
          "bContext": "Claude Code · effort high · eight hard validated tasks, effort ladder",
          "calculation": true
        },
        {
          "metric": "Reasoning cost per strict pass by effort, with the total (calculation) (Reasoning cost per strict pass)",
          "aValue": 0.004317,
          "bValue": 0.009369,
          "unit": "usd",
          "aDisplay": "$0.0043",
          "bDisplay": "$0.0094",
          "winner": "unclear",
          "basis": "More or fewer usd is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "n": 16,
          "chartId": "thinking-bill-by-effort",
          "aN": 16,
          "bN": 16,
          "aContext": "Claude Code · effort low",
          "bContext": "Claude Code · effort high",
          "calculation": true
        },
        {
          "metric": "Reasoning cost per strict pass by effort, with the total (calculation) (Total cost per strict pass)",
          "aValue": 0.012191,
          "bValue": 0.016705,
          "unit": "usd",
          "aDisplay": "$0.012",
          "bDisplay": "$0.017",
          "winner": "unclear",
          "basis": "More or fewer usd is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "n": 16,
          "chartId": "thinking-bill-by-effort",
          "aN": 16,
          "bN": 16,
          "aContext": "Claude Code · effort low",
          "bContext": "Claude Code · effort high",
          "calculation": true
        }
      ]
    },
    {
      "slug": "claude-sonnet-5-5-low-vs-default",
      "entity": "claude-sonnet-5-5",
      "aEffort": "low",
      "bEffort": "default",
      "title": "Claude Sonnet 5.5: low vs default effort",
      "seoTitle": "Claude Sonnet 5.5: low vs default effort, measured",
      "description": "Claude Sonnet 5.5 at low vs default effort: 3 measured metrics from one study, with sample sizes, intervals and every failure counted.",
      "verdict": "Claude Sonnet 5.5 at low effort and Claude Sonnet 5.5 at default effort share 3 measured metrics and 3 list-price calculations from 2 studies. Only the effort setting differs between the two sides of a row; the route and the task set are the same. \"default\" means the effort flag was not passed, so its level is the CLI’s choice. No row separates them: every interval or run range overlaps, too few runs were recorded, no interval was recorded, or more is not better for that metric. The rows are 1 tie and 5 unclear; each row says why. Calculation rows are derived from list prices and recorded counts; they are not bills or runs.",
      "rows": [
        {
          "metric": "Strict pass rate by effort on eight hard tasks",
          "aValue": 1,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "100% (16/16)",
          "bDisplay": "100% (16/16)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Sonnet 5.5 at low effort 81% to 100%; Claude Sonnet 5.5 at default effort 81% to 100%), so this sample cannot separate them.",
          "studySlug": "effort-ladder",
          "n": 16,
          "chartId": "effort-ladder-pass-rate",
          "aN": 16,
          "bN": 16,
          "aContext": "Claude Code · effort low · eight hard validated tasks, effort ladder",
          "bContext": "Claude Code · eight hard validated tasks, effort ladder",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.8064,
            1
          ],
          "bRange": [
            0.8064,
            1
          ]
        },
        {
          "metric": "Total time per call by effort on hard tasks",
          "aValue": 5.82,
          "bValue": 7.97,
          "unit": "seconds",
          "aDisplay": "5.82 s",
          "bDisplay": "7.97 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Sonnet 5.5 at low effort 2.78 s to 20.0 s; Claude Sonnet 5.5 at default effort 2.26 s to 21.6 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "effort-ladder",
          "n": 16,
          "chartId": "effort-ladder-total-latency",
          "aN": 16,
          "bN": 16,
          "aContext": "Claude Code · effort low · eight hard validated tasks, effort ladder",
          "bContext": "Claude Code · eight hard validated tasks, effort ladder",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            2.78,
            19.96
          ],
          "bRange": [
            2.26,
            21.61
          ]
        },
        {
          "metric": "Output tokens per call by effort on hard tasks (Output tokens)",
          "aValue": 667,
          "bValue": 1054,
          "unit": "tokens",
          "aDisplay": "667",
          "bDisplay": "1,054",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "effort-ladder",
          "n": 16,
          "chartId": "effort-ladder-output-tokens",
          "aN": 16,
          "bN": 16,
          "aContext": "Claude Code · effort low · eight hard validated tasks, effort ladder",
          "bContext": "Claude Code · eight hard validated tasks, effort ladder"
        },
        {
          "metric": "List-price cost per strict pass by effort (calculation)",
          "aValue": 0.01219,
          "bValue": 0.01398,
          "unit": "usd",
          "aDisplay": "$0.012",
          "bDisplay": "$0.014",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.012 vs $0.014) is not tested against run-to-run variation.",
          "studySlug": "effort-ladder",
          "n": 16,
          "chartId": "effort-ladder-cost-per-pass",
          "aN": 16,
          "bN": 16,
          "aContext": "Claude Code · effort low · eight hard validated tasks, effort ladder",
          "bContext": "Claude Code · eight hard validated tasks, effort ladder",
          "calculation": true
        },
        {
          "metric": "Reasoning cost per strict pass by effort, with the total (calculation) (Reasoning cost per strict pass)",
          "aValue": 0.004317,
          "bValue": 0.006299,
          "unit": "usd",
          "aDisplay": "$0.0043",
          "bDisplay": "$0.0063",
          "winner": "unclear",
          "basis": "More or fewer usd is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "n": 16,
          "chartId": "thinking-bill-by-effort",
          "aN": 16,
          "bN": 16,
          "aContext": "Claude Code · effort low",
          "bContext": "Claude Code",
          "calculation": true
        },
        {
          "metric": "Reasoning cost per strict pass by effort, with the total (calculation) (Total cost per strict pass)",
          "aValue": 0.012191,
          "bValue": 0.013978,
          "unit": "usd",
          "aDisplay": "$0.012",
          "bDisplay": "$0.014",
          "winner": "unclear",
          "basis": "More or fewer usd is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "n": 16,
          "chartId": "thinking-bill-by-effort",
          "aN": 16,
          "bN": 16,
          "aContext": "Claude Code · effort low",
          "bContext": "Claude Code",
          "calculation": true
        }
      ]
    },
    {
      "slug": "claude-sonnet-5-5-medium-vs-high",
      "entity": "claude-sonnet-5-5",
      "aEffort": "medium",
      "bEffort": "high",
      "title": "Claude Sonnet 5.5: medium vs high effort",
      "seoTitle": "Claude Sonnet 5.5: medium vs high effort, measured",
      "description": "Claude Sonnet 5.5 at medium vs high effort: 3 measured metrics from one study, with sample sizes, intervals and every failure counted.",
      "verdict": "Claude Sonnet 5.5 at medium effort and Claude Sonnet 5.5 at high effort share 3 measured metrics and 3 list-price calculations from 2 studies. Only the effort setting differs between the two sides of a row; the route and the task set are the same. No row separates them: every interval or run range overlaps, too few runs were recorded, no interval was recorded, or more is not better for that metric. The rows are 1 tie and 5 unclear; each row says why. Calculation rows are derived from list prices and recorded counts; they are not bills or runs.",
      "rows": [
        {
          "metric": "Strict pass rate by effort on eight hard tasks",
          "aValue": 1,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "100% (16/16)",
          "bDisplay": "100% (16/16)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Sonnet 5.5 at medium effort 81% to 100%; Claude Sonnet 5.5 at high effort 81% to 100%), so this sample cannot separate them.",
          "studySlug": "effort-ladder",
          "n": 16,
          "chartId": "effort-ladder-pass-rate",
          "aN": 16,
          "bN": 16,
          "aContext": "Claude Code · effort medium · eight hard validated tasks, effort ladder",
          "bContext": "Claude Code · effort high · eight hard validated tasks, effort ladder",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.8064,
            1
          ],
          "bRange": [
            0.8064,
            1
          ]
        },
        {
          "metric": "Total time per call by effort on hard tasks",
          "aValue": 7.63,
          "bValue": 8.81,
          "unit": "seconds",
          "aDisplay": "7.63 s",
          "bDisplay": "8.81 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Sonnet 5.5 at medium effort 2.71 s to 24.0 s; Claude Sonnet 5.5 at high effort 2.93 s to 35.8 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "effort-ladder",
          "n": 16,
          "chartId": "effort-ladder-total-latency",
          "aN": 16,
          "bN": 16,
          "aContext": "Claude Code · effort medium · eight hard validated tasks, effort ladder",
          "bContext": "Claude Code · effort high · eight hard validated tasks, effort ladder",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            2.71,
            24.01
          ],
          "bRange": [
            2.93,
            35.81
          ]
        },
        {
          "metric": "Output tokens per call by effort on hard tasks (Output tokens)",
          "aValue": 770,
          "bValue": 1192,
          "unit": "tokens",
          "aDisplay": "770",
          "bDisplay": "1,192",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "effort-ladder",
          "n": 16,
          "chartId": "effort-ladder-output-tokens",
          "aN": 16,
          "bN": 16,
          "aContext": "Claude Code · effort medium · eight hard validated tasks, effort ladder",
          "bContext": "Claude Code · effort high · eight hard validated tasks, effort ladder"
        },
        {
          "metric": "List-price cost per strict pass by effort (calculation)",
          "aValue": 0.01352,
          "bValue": 0.01671,
          "unit": "usd",
          "aDisplay": "$0.014",
          "bDisplay": "$0.017",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.014 vs $0.017) is not tested against run-to-run variation.",
          "studySlug": "effort-ladder",
          "n": 16,
          "chartId": "effort-ladder-cost-per-pass",
          "aN": 16,
          "bN": 16,
          "aContext": "Claude Code · effort medium · eight hard validated tasks, effort ladder",
          "bContext": "Claude Code · effort high · eight hard validated tasks, effort ladder",
          "calculation": true
        },
        {
          "metric": "Reasoning cost per strict pass by effort, with the total (calculation) (Reasoning cost per strict pass)",
          "aValue": 0.005946,
          "bValue": 0.009369,
          "unit": "usd",
          "aDisplay": "$0.0059",
          "bDisplay": "$0.0094",
          "winner": "unclear",
          "basis": "More or fewer usd is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "n": 16,
          "chartId": "thinking-bill-by-effort",
          "aN": 16,
          "bN": 16,
          "aContext": "Claude Code · effort medium",
          "bContext": "Claude Code · effort high",
          "calculation": true
        },
        {
          "metric": "Reasoning cost per strict pass by effort, with the total (calculation) (Total cost per strict pass)",
          "aValue": 0.01352,
          "bValue": 0.016705,
          "unit": "usd",
          "aDisplay": "$0.014",
          "bDisplay": "$0.017",
          "winner": "unclear",
          "basis": "More or fewer usd is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "n": 16,
          "chartId": "thinking-bill-by-effort",
          "aN": 16,
          "bN": 16,
          "aContext": "Claude Code · effort medium",
          "bContext": "Claude Code · effort high",
          "calculation": true
        }
      ]
    },
    {
      "slug": "claude-sonnet-5-5-medium-vs-default",
      "entity": "claude-sonnet-5-5",
      "aEffort": "medium",
      "bEffort": "default",
      "title": "Claude Sonnet 5.5: medium vs default effort",
      "seoTitle": "Claude Sonnet 5.5: medium vs default effort, measured",
      "description": "Claude Sonnet 5.5 at medium vs default effort: 3 measured metrics from one study, with sample sizes, intervals and every failure counted.",
      "verdict": "Claude Sonnet 5.5 at medium effort and Claude Sonnet 5.5 at default effort share 3 measured metrics and 3 list-price calculations from 2 studies. Only the effort setting differs between the two sides of a row; the route and the task set are the same. \"default\" means the effort flag was not passed, so its level is the CLI’s choice. No row separates them: every interval or run range overlaps, too few runs were recorded, no interval was recorded, or more is not better for that metric. The rows are 1 tie and 5 unclear; each row says why. Calculation rows are derived from list prices and recorded counts; they are not bills or runs.",
      "rows": [
        {
          "metric": "Strict pass rate by effort on eight hard tasks",
          "aValue": 1,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "100% (16/16)",
          "bDisplay": "100% (16/16)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Sonnet 5.5 at medium effort 81% to 100%; Claude Sonnet 5.5 at default effort 81% to 100%), so this sample cannot separate them.",
          "studySlug": "effort-ladder",
          "n": 16,
          "chartId": "effort-ladder-pass-rate",
          "aN": 16,
          "bN": 16,
          "aContext": "Claude Code · effort medium · eight hard validated tasks, effort ladder",
          "bContext": "Claude Code · eight hard validated tasks, effort ladder",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.8064,
            1
          ],
          "bRange": [
            0.8064,
            1
          ]
        },
        {
          "metric": "Total time per call by effort on hard tasks",
          "aValue": 7.63,
          "bValue": 7.97,
          "unit": "seconds",
          "aDisplay": "7.63 s",
          "bDisplay": "7.97 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Sonnet 5.5 at medium effort 2.71 s to 24.0 s; Claude Sonnet 5.5 at default effort 2.26 s to 21.6 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "effort-ladder",
          "n": 16,
          "chartId": "effort-ladder-total-latency",
          "aN": 16,
          "bN": 16,
          "aContext": "Claude Code · effort medium · eight hard validated tasks, effort ladder",
          "bContext": "Claude Code · eight hard validated tasks, effort ladder",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            2.71,
            24.01
          ],
          "bRange": [
            2.26,
            21.61
          ]
        },
        {
          "metric": "Output tokens per call by effort on hard tasks (Output tokens)",
          "aValue": 770,
          "bValue": 1054,
          "unit": "tokens",
          "aDisplay": "770",
          "bDisplay": "1,054",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "effort-ladder",
          "n": 16,
          "chartId": "effort-ladder-output-tokens",
          "aN": 16,
          "bN": 16,
          "aContext": "Claude Code · effort medium · eight hard validated tasks, effort ladder",
          "bContext": "Claude Code · eight hard validated tasks, effort ladder"
        },
        {
          "metric": "List-price cost per strict pass by effort (calculation)",
          "aValue": 0.01352,
          "bValue": 0.01398,
          "unit": "usd",
          "aDisplay": "$0.014",
          "bDisplay": "$0.014",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.014 vs $0.014) is not tested against run-to-run variation.",
          "studySlug": "effort-ladder",
          "n": 16,
          "chartId": "effort-ladder-cost-per-pass",
          "aN": 16,
          "bN": 16,
          "aContext": "Claude Code · effort medium · eight hard validated tasks, effort ladder",
          "bContext": "Claude Code · eight hard validated tasks, effort ladder",
          "calculation": true
        },
        {
          "metric": "Reasoning cost per strict pass by effort, with the total (calculation) (Reasoning cost per strict pass)",
          "aValue": 0.005946,
          "bValue": 0.006299,
          "unit": "usd",
          "aDisplay": "$0.0059",
          "bDisplay": "$0.0063",
          "winner": "unclear",
          "basis": "More or fewer usd is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "n": 16,
          "chartId": "thinking-bill-by-effort",
          "aN": 16,
          "bN": 16,
          "aContext": "Claude Code · effort medium",
          "bContext": "Claude Code",
          "calculation": true
        },
        {
          "metric": "Reasoning cost per strict pass by effort, with the total (calculation) (Total cost per strict pass)",
          "aValue": 0.01352,
          "bValue": 0.013978,
          "unit": "usd",
          "aDisplay": "$0.014",
          "bDisplay": "$0.014",
          "winner": "unclear",
          "basis": "More or fewer usd is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "n": 16,
          "chartId": "thinking-bill-by-effort",
          "aN": 16,
          "bN": 16,
          "aContext": "Claude Code · effort medium",
          "bContext": "Claude Code",
          "calculation": true
        }
      ]
    },
    {
      "slug": "claude-sonnet-5-5-high-vs-default",
      "entity": "claude-sonnet-5-5",
      "aEffort": "high",
      "bEffort": "default",
      "title": "Claude Sonnet 5.5: high vs default effort",
      "seoTitle": "Claude Sonnet 5.5: high vs default effort, measured",
      "description": "Claude Sonnet 5.5 at high vs default effort: 3 measured metrics from one study, with sample sizes, intervals and every failure counted.",
      "verdict": "Claude Sonnet 5.5 at high effort and Claude Sonnet 5.5 at default effort share 3 measured metrics and 3 list-price calculations from 2 studies. Only the effort setting differs between the two sides of a row; the route and the task set are the same. \"default\" means the effort flag was not passed, so its level is the CLI’s choice. No row separates them: every interval or run range overlaps, too few runs were recorded, no interval was recorded, or more is not better for that metric. The rows are 1 tie and 5 unclear; each row says why. Calculation rows are derived from list prices and recorded counts; they are not bills or runs.",
      "rows": [
        {
          "metric": "Strict pass rate by effort on eight hard tasks",
          "aValue": 1,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "100% (16/16)",
          "bDisplay": "100% (16/16)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Sonnet 5.5 at high effort 81% to 100%; Claude Sonnet 5.5 at default effort 81% to 100%), so this sample cannot separate them.",
          "studySlug": "effort-ladder",
          "n": 16,
          "chartId": "effort-ladder-pass-rate",
          "aN": 16,
          "bN": 16,
          "aContext": "Claude Code · effort high · eight hard validated tasks, effort ladder",
          "bContext": "Claude Code · eight hard validated tasks, effort ladder",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.8064,
            1
          ],
          "bRange": [
            0.8064,
            1
          ]
        },
        {
          "metric": "Total time per call by effort on hard tasks",
          "aValue": 8.81,
          "bValue": 7.97,
          "unit": "seconds",
          "aDisplay": "8.81 s",
          "bDisplay": "7.97 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Sonnet 5.5 at high effort 2.93 s to 35.8 s; Claude Sonnet 5.5 at default effort 2.26 s to 21.6 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "effort-ladder",
          "n": 16,
          "chartId": "effort-ladder-total-latency",
          "aN": 16,
          "bN": 16,
          "aContext": "Claude Code · effort high · eight hard validated tasks, effort ladder",
          "bContext": "Claude Code · eight hard validated tasks, effort ladder",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            2.93,
            35.81
          ],
          "bRange": [
            2.26,
            21.61
          ]
        },
        {
          "metric": "Output tokens per call by effort on hard tasks (Output tokens)",
          "aValue": 1192,
          "bValue": 1054,
          "unit": "tokens",
          "aDisplay": "1,192",
          "bDisplay": "1,054",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "effort-ladder",
          "n": 16,
          "chartId": "effort-ladder-output-tokens",
          "aN": 16,
          "bN": 16,
          "aContext": "Claude Code · effort high · eight hard validated tasks, effort ladder",
          "bContext": "Claude Code · eight hard validated tasks, effort ladder"
        },
        {
          "metric": "List-price cost per strict pass by effort (calculation)",
          "aValue": 0.01671,
          "bValue": 0.01398,
          "unit": "usd",
          "aDisplay": "$0.017",
          "bDisplay": "$0.014",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.017 vs $0.014) is not tested against run-to-run variation.",
          "studySlug": "effort-ladder",
          "n": 16,
          "chartId": "effort-ladder-cost-per-pass",
          "aN": 16,
          "bN": 16,
          "aContext": "Claude Code · effort high · eight hard validated tasks, effort ladder",
          "bContext": "Claude Code · eight hard validated tasks, effort ladder",
          "calculation": true
        },
        {
          "metric": "Reasoning cost per strict pass by effort, with the total (calculation) (Reasoning cost per strict pass)",
          "aValue": 0.009369,
          "bValue": 0.006299,
          "unit": "usd",
          "aDisplay": "$0.0094",
          "bDisplay": "$0.0063",
          "winner": "unclear",
          "basis": "More or fewer usd is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "n": 16,
          "chartId": "thinking-bill-by-effort",
          "aN": 16,
          "bN": 16,
          "aContext": "Claude Code · effort high",
          "bContext": "Claude Code",
          "calculation": true
        },
        {
          "metric": "Reasoning cost per strict pass by effort, with the total (calculation) (Total cost per strict pass)",
          "aValue": 0.016705,
          "bValue": 0.013978,
          "unit": "usd",
          "aDisplay": "$0.017",
          "bDisplay": "$0.014",
          "winner": "unclear",
          "basis": "More or fewer usd is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "n": 16,
          "chartId": "thinking-bill-by-effort",
          "aN": 16,
          "bN": 16,
          "aContext": "Claude Code · effort high",
          "bContext": "Claude Code",
          "calculation": true
        }
      ]
    },
    {
      "slug": "claude-opus-5-5-low-vs-medium",
      "entity": "claude-opus-5-5",
      "aEffort": "low",
      "bEffort": "medium",
      "title": "Claude Opus 5.5: low vs medium effort",
      "seoTitle": "Claude Opus 5.5: low vs medium effort, measured",
      "description": "Claude Opus 5.5 at low vs medium effort: 3 measured metrics from one study, with sample sizes, intervals and every failure counted.",
      "verdict": "Claude Opus 5.5 at low effort and Claude Opus 5.5 at medium effort share 3 measured metrics and 3 list-price calculations from 2 studies. Only the effort setting differs between the two sides of a row; the route and the task set are the same. No row separates them: every interval or run range overlaps, too few runs were recorded, no interval was recorded, or more is not better for that metric. The rows are 1 tie and 5 unclear; each row says why. Calculation rows are derived from list prices and recorded counts; they are not bills or runs.",
      "rows": [
        {
          "metric": "Strict pass rate by effort on eight hard tasks",
          "aValue": 1,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "100% (16/16)",
          "bDisplay": "100% (16/16)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Opus 5.5 at low effort 81% to 100%; Claude Opus 5.5 at medium effort 81% to 100%), so this sample cannot separate them.",
          "studySlug": "effort-ladder",
          "n": 16,
          "chartId": "effort-ladder-pass-rate",
          "aN": 16,
          "bN": 16,
          "aContext": "Claude Code · effort low · eight hard validated tasks, effort ladder",
          "bContext": "Claude Code · effort medium · eight hard validated tasks, effort ladder",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.8064,
            1
          ],
          "bRange": [
            0.8064,
            1
          ]
        },
        {
          "metric": "Total time per call by effort on hard tasks",
          "aValue": 7.5,
          "bValue": 9.72,
          "unit": "seconds",
          "aDisplay": "7.50 s",
          "bDisplay": "9.72 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Opus 5.5 at low effort 3.34 s to 15.8 s; Claude Opus 5.5 at medium effort 4.78 s to 31.4 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "effort-ladder",
          "n": 16,
          "chartId": "effort-ladder-total-latency",
          "aN": 16,
          "bN": 16,
          "aContext": "Claude Code · effort low · eight hard validated tasks, effort ladder",
          "bContext": "Claude Code · effort medium · eight hard validated tasks, effort ladder",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            3.34,
            15.82
          ],
          "bRange": [
            4.78,
            31.36
          ]
        },
        {
          "metric": "Output tokens per call by effort on hard tasks (Output tokens)",
          "aValue": 594,
          "bValue": 853,
          "unit": "tokens",
          "aDisplay": "594",
          "bDisplay": "853",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "effort-ladder",
          "n": 16,
          "chartId": "effort-ladder-output-tokens",
          "aN": 16,
          "bN": 16,
          "aContext": "Claude Code · effort low · eight hard validated tasks, effort ladder",
          "bContext": "Claude Code · effort medium · eight hard validated tasks, effort ladder"
        },
        {
          "metric": "List-price cost per strict pass by effort (calculation)",
          "aValue": 0.02115,
          "bValue": 0.02947,
          "unit": "usd",
          "aDisplay": "$0.021",
          "bDisplay": "$0.029",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.021 vs $0.029) is not tested against run-to-run variation.",
          "studySlug": "effort-ladder",
          "n": 16,
          "chartId": "effort-ladder-cost-per-pass",
          "aN": 16,
          "bN": 16,
          "aContext": "Claude Code · effort low · eight hard validated tasks, effort ladder",
          "bContext": "Claude Code · effort medium · eight hard validated tasks, effort ladder",
          "calculation": true
        },
        {
          "metric": "Reasoning cost per strict pass by effort, with the total (calculation) (Reasoning cost per strict pass)",
          "aValue": 0.005031,
          "bValue": 0.01344,
          "unit": "usd",
          "aDisplay": "$0.0050",
          "bDisplay": "$0.013",
          "winner": "unclear",
          "basis": "More or fewer usd is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "n": 16,
          "chartId": "thinking-bill-by-effort",
          "aN": 16,
          "bN": 16,
          "aContext": "Claude Code · effort low",
          "bContext": "Claude Code · effort medium",
          "calculation": true
        },
        {
          "metric": "Reasoning cost per strict pass by effort, with the total (calculation) (Total cost per strict pass)",
          "aValue": 0.021152,
          "bValue": 0.029475,
          "unit": "usd",
          "aDisplay": "$0.021",
          "bDisplay": "$0.029",
          "winner": "unclear",
          "basis": "More or fewer usd is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "n": 16,
          "chartId": "thinking-bill-by-effort",
          "aN": 16,
          "bN": 16,
          "aContext": "Claude Code · effort low",
          "bContext": "Claude Code · effort medium",
          "calculation": true
        }
      ]
    },
    {
      "slug": "claude-opus-5-5-low-vs-high",
      "entity": "claude-opus-5-5",
      "aEffort": "low",
      "bEffort": "high",
      "title": "Claude Opus 5.5: low vs high effort",
      "seoTitle": "Claude Opus 5.5: low vs high effort, measured",
      "description": "Claude Opus 5.5 at low vs high effort: 9 measured metrics from 2 studies, with sample sizes, intervals and every failure counted.",
      "verdict": "Claude Opus 5.5 at low effort and Claude Opus 5.5 at high effort share 9 measured metrics and 5 list-price calculations from 3 studies. Only the effort setting differs between the two sides of a row; the route and the task set are the same. No row separates them: every interval or run range overlaps, too few runs were recorded, no interval was recorded, or more is not better for that metric. The rows are 3 ties and 11 unclear; each row says why. Calculation rows are derived from list prices and recorded counts; they are not bills or runs. Some rows rest on small samples (n = 15 at the smallest).",
      "rows": [
        {
          "metric": "Pass rate on five validated tasks",
          "aValue": 1,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "100% (15/15)",
          "bDisplay": "100% (15/15)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Opus 5.5 at low effort 80% to 100%; Claude Opus 5.5 at high effort 80% to 100%), so this sample cannot separate them.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-pass-rate",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code · effort low · five short validated tasks",
          "bContext": "Claude Code · effort high · five short validated tasks",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.7961,
            1
          ],
          "bRange": [
            0.7961,
            1
          ]
        },
        {
          "metric": "Total time per call",
          "aValue": 2.83,
          "bValue": 2.71,
          "unit": "seconds",
          "aDisplay": "2.83 s",
          "bDisplay": "2.71 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Opus 5.5 at low effort 2.35 s to 6.62 s; Claude Opus 5.5 at high effort 2.45 s to 11.8 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-total-latency",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code · effort low · five short validated tasks",
          "bContext": "Claude Code · effort high · five short validated tasks",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            2.35,
            6.62
          ],
          "bRange": [
            2.45,
            11.78
          ]
        },
        {
          "metric": "Time to first useful output",
          "aValue": 2.39,
          "bValue": 2.04,
          "unit": "seconds",
          "aDisplay": "2.39 s",
          "bDisplay": "2.04 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Opus 5.5 at low effort 1.45 s to 4.90 s; Claude Opus 5.5 at high effort 1.40 s to 9.94 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-first-useful-latency",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code · effort low · five short validated tasks",
          "bContext": "Claude Code · effort high · five short validated tasks",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            1.45,
            4.9
          ],
          "bRange": [
            1.4,
            9.94
          ]
        },
        {
          "metric": "Input tokens per call: what the CLI sends (Cache read)",
          "aValue": 1463,
          "bValue": 1463,
          "unit": "tokens",
          "aDisplay": "1,463",
          "bDisplay": "1,463",
          "winner": "tie",
          "basis": "Same value. More or fewer is not better by itself for this metric.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-input-tokens",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code · effort low · five short validated tasks",
          "bContext": "Claude Code · effort high · five short validated tasks"
        },
        {
          "metric": "Input tokens per call: what the CLI sends (Other input)",
          "aValue": 618,
          "bValue": 619,
          "unit": "tokens",
          "aDisplay": "618",
          "bDisplay": "619",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-input-tokens",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code · effort low · five short validated tasks",
          "bContext": "Claude Code · effort high · five short validated tasks"
        },
        {
          "metric": "Output tokens per call (Output tokens)",
          "aValue": 64,
          "bValue": 78,
          "unit": "tokens",
          "aDisplay": "64",
          "bDisplay": "78",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-output-tokens",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code · effort low · five short validated tasks",
          "bContext": "Claude Code · effort high · five short validated tasks"
        },
        {
          "metric": "List-price cost per call (calculation)",
          "aValue": 0.00688,
          "bValue": 0.00694,
          "unit": "usd",
          "aDisplay": "$0.0069",
          "bDisplay": "$0.0069",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Opus 5.5 at low effort $0.0058 to $0.018; Claude Opus 5.5 at high effort $0.0059 to $0.027); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-list-price-per-call",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code · effort low · five short validated tasks",
          "bContext": "Claude Code · effort high · five short validated tasks",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            0.00582,
            0.01793
          ],
          "bRange": [
            0.00592,
            0.02708
          ],
          "calculation": true
        },
        {
          "metric": "List-price cost per passing answer (calculation)",
          "aValue": 0.00829,
          "bValue": 0.01049,
          "unit": "usd",
          "aDisplay": "$0.0083",
          "bDisplay": "$0.010",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.0083 vs $0.010) is not tested against run-to-run variation.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-cost-per-pass",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code · effort low · five short validated tasks",
          "bContext": "Claude Code · effort high · five short validated tasks",
          "calculation": true
        },
        {
          "metric": "Strict pass rate by effort on eight hard tasks",
          "aValue": 1,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "100% (16/16)",
          "bDisplay": "100% (16/16)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Opus 5.5 at low effort 81% to 100%; Claude Opus 5.5 at high effort 81% to 100%), so this sample cannot separate them.",
          "studySlug": "effort-ladder",
          "n": 16,
          "chartId": "effort-ladder-pass-rate",
          "aN": 16,
          "bN": 16,
          "aContext": "Claude Code · effort low · eight hard validated tasks, effort ladder",
          "bContext": "Claude Code · effort high · eight hard validated tasks, effort ladder",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.8064,
            1
          ],
          "bRange": [
            0.8064,
            1
          ]
        },
        {
          "metric": "Total time per call by effort on hard tasks",
          "aValue": 7.5,
          "bValue": 10.11,
          "unit": "seconds",
          "aDisplay": "7.50 s",
          "bDisplay": "10.1 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Opus 5.5 at low effort 3.34 s to 15.8 s; Claude Opus 5.5 at high effort 3.63 s to 63.0 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "effort-ladder",
          "n": 16,
          "chartId": "effort-ladder-total-latency",
          "aN": 16,
          "bN": 16,
          "aContext": "Claude Code · effort low · eight hard validated tasks, effort ladder",
          "bContext": "Claude Code · effort high · eight hard validated tasks, effort ladder",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            3.34,
            15.82
          ],
          "bRange": [
            3.63,
            63
          ]
        },
        {
          "metric": "Output tokens per call by effort on hard tasks (Output tokens)",
          "aValue": 594,
          "bValue": 1052,
          "unit": "tokens",
          "aDisplay": "594",
          "bDisplay": "1,052",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "effort-ladder",
          "n": 16,
          "chartId": "effort-ladder-output-tokens",
          "aN": 16,
          "bN": 16,
          "aContext": "Claude Code · effort low · eight hard validated tasks, effort ladder",
          "bContext": "Claude Code · effort high · eight hard validated tasks, effort ladder"
        },
        {
          "metric": "List-price cost per strict pass by effort (calculation)",
          "aValue": 0.02115,
          "bValue": 0.03368,
          "unit": "usd",
          "aDisplay": "$0.021",
          "bDisplay": "$0.034",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.021 vs $0.034) is not tested against run-to-run variation.",
          "studySlug": "effort-ladder",
          "n": 16,
          "chartId": "effort-ladder-cost-per-pass",
          "aN": 16,
          "bN": 16,
          "aContext": "Claude Code · effort low · eight hard validated tasks, effort ladder",
          "bContext": "Claude Code · effort high · eight hard validated tasks, effort ladder",
          "calculation": true
        },
        {
          "metric": "Reasoning cost per strict pass by effort, with the total (calculation) (Reasoning cost per strict pass)",
          "aValue": 0.005031,
          "bValue": 0.018034,
          "unit": "usd",
          "aDisplay": "$0.0050",
          "bDisplay": "$0.018",
          "winner": "unclear",
          "basis": "More or fewer usd is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "n": 16,
          "chartId": "thinking-bill-by-effort",
          "aN": 16,
          "bN": 16,
          "aContext": "Claude Code · effort low",
          "bContext": "Claude Code · effort high",
          "calculation": true
        },
        {
          "metric": "Reasoning cost per strict pass by effort, with the total (calculation) (Total cost per strict pass)",
          "aValue": 0.021152,
          "bValue": 0.033677,
          "unit": "usd",
          "aDisplay": "$0.021",
          "bDisplay": "$0.034",
          "winner": "unclear",
          "basis": "More or fewer usd is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "n": 16,
          "chartId": "thinking-bill-by-effort",
          "aN": 16,
          "bN": 16,
          "aContext": "Claude Code · effort low",
          "bContext": "Claude Code · effort high",
          "calculation": true
        }
      ]
    },
    {
      "slug": "claude-opus-5-5-low-vs-default",
      "entity": "claude-opus-5-5",
      "aEffort": "low",
      "bEffort": "default",
      "title": "Claude Opus 5.5: low vs default effort",
      "seoTitle": "Claude Opus 5.5: low vs default effort, measured",
      "description": "Claude Opus 5.5 at low vs default effort: 9 measured metrics from 2 studies, with sample sizes, intervals and every failure counted.",
      "verdict": "Claude Opus 5.5 at low effort and Claude Opus 5.5 at default effort share 9 measured metrics and 5 list-price calculations from 3 studies. Only the effort setting differs between the two sides of a row; the route and the task set are the same. \"default\" means the effort flag was not passed, so its level is the CLI’s choice. No row separates them: every interval or run range overlaps, too few runs were recorded, no interval was recorded, or more is not better for that metric. The rows are 3 ties and 11 unclear; each row says why. Calculation rows are derived from list prices and recorded counts; they are not bills or runs. Some rows rest on small samples (n = 15 at the smallest).",
      "rows": [
        {
          "metric": "Pass rate on five validated tasks",
          "aValue": 1,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "100% (15/15)",
          "bDisplay": "100% (15/15)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Opus 5.5 at low effort 80% to 100%; Claude Opus 5.5 at default effort 80% to 100%), so this sample cannot separate them.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-pass-rate",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code · effort low · five short validated tasks",
          "bContext": "Claude Code · five short validated tasks",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.7961,
            1
          ],
          "bRange": [
            0.7961,
            1
          ]
        },
        {
          "metric": "Total time per call",
          "aValue": 2.83,
          "bValue": 2.75,
          "unit": "seconds",
          "aDisplay": "2.83 s",
          "bDisplay": "2.75 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Opus 5.5 at low effort 2.35 s to 6.62 s; Claude Opus 5.5 at default effort 2.47 s to 8.91 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-total-latency",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code · effort low · five short validated tasks",
          "bContext": "Claude Code · five short validated tasks",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            2.35,
            6.62
          ],
          "bRange": [
            2.47,
            8.91
          ]
        },
        {
          "metric": "Time to first useful output",
          "aValue": 2.39,
          "bValue": 1.92,
          "unit": "seconds",
          "aDisplay": "2.39 s",
          "bDisplay": "1.92 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Opus 5.5 at low effort 1.45 s to 4.90 s; Claude Opus 5.5 at default effort 1.56 s to 7.23 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-first-useful-latency",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code · effort low · five short validated tasks",
          "bContext": "Claude Code · five short validated tasks",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            1.45,
            4.9
          ],
          "bRange": [
            1.56,
            7.23
          ]
        },
        {
          "metric": "Input tokens per call: what the CLI sends (Cache read)",
          "aValue": 1463,
          "bValue": 1401,
          "unit": "tokens",
          "aDisplay": "1,463",
          "bDisplay": "1,401",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-input-tokens",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code · effort low · five short validated tasks",
          "bContext": "Claude Code · five short validated tasks"
        },
        {
          "metric": "Input tokens per call: what the CLI sends (Other input)",
          "aValue": 618,
          "bValue": 680,
          "unit": "tokens",
          "aDisplay": "618",
          "bDisplay": "680",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-input-tokens",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code · effort low · five short validated tasks",
          "bContext": "Claude Code · five short validated tasks"
        },
        {
          "metric": "Output tokens per call (Output tokens)",
          "aValue": 64,
          "bValue": 64,
          "unit": "tokens",
          "aDisplay": "64",
          "bDisplay": "64",
          "winner": "tie",
          "basis": "Same value. More or fewer is not better by itself for this metric.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-output-tokens",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code · effort low · five short validated tasks",
          "bContext": "Claude Code · five short validated tasks"
        },
        {
          "metric": "List-price cost per call (calculation)",
          "aValue": 0.00688,
          "bValue": 0.00688,
          "unit": "usd",
          "aDisplay": "$0.0069",
          "bDisplay": "$0.0069",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Opus 5.5 at low effort $0.0058 to $0.018; Claude Opus 5.5 at default effort $0.0059 to $0.022); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-list-price-per-call",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code · effort low · five short validated tasks",
          "bContext": "Claude Code · five short validated tasks",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            0.00582,
            0.01793
          ],
          "bRange": [
            0.00592,
            0.02226
          ],
          "calculation": true
        },
        {
          "metric": "List-price cost per passing answer (calculation)",
          "aValue": 0.00829,
          "bValue": 0.01009,
          "unit": "usd",
          "aDisplay": "$0.0083",
          "bDisplay": "$0.010",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.0083 vs $0.010) is not tested against run-to-run variation.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-cost-per-pass",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code · effort low · five short validated tasks",
          "bContext": "Claude Code · five short validated tasks",
          "calculation": true
        },
        {
          "metric": "Strict pass rate by effort on eight hard tasks",
          "aValue": 1,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "100% (16/16)",
          "bDisplay": "100% (16/16)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Opus 5.5 at low effort 81% to 100%; Claude Opus 5.5 at default effort 81% to 100%), so this sample cannot separate them.",
          "studySlug": "effort-ladder",
          "n": 16,
          "chartId": "effort-ladder-pass-rate",
          "aN": 16,
          "bN": 16,
          "aContext": "Claude Code · effort low · eight hard validated tasks, effort ladder",
          "bContext": "Claude Code · eight hard validated tasks, effort ladder",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.8064,
            1
          ],
          "bRange": [
            0.8064,
            1
          ]
        },
        {
          "metric": "Total time per call by effort on hard tasks",
          "aValue": 7.5,
          "bValue": 9.18,
          "unit": "seconds",
          "aDisplay": "7.50 s",
          "bDisplay": "9.18 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Opus 5.5 at low effort 3.34 s to 15.8 s; Claude Opus 5.5 at default effort 4.24 s to 27.2 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "effort-ladder",
          "n": 16,
          "chartId": "effort-ladder-total-latency",
          "aN": 16,
          "bN": 16,
          "aContext": "Claude Code · effort low · eight hard validated tasks, effort ladder",
          "bContext": "Claude Code · eight hard validated tasks, effort ladder",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            3.34,
            15.82
          ],
          "bRange": [
            4.24,
            27.21
          ]
        },
        {
          "metric": "Output tokens per call by effort on hard tasks (Output tokens)",
          "aValue": 594,
          "bValue": 945,
          "unit": "tokens",
          "aDisplay": "594",
          "bDisplay": "945",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "effort-ladder",
          "n": 16,
          "chartId": "effort-ladder-output-tokens",
          "aN": 16,
          "bN": 16,
          "aContext": "Claude Code · effort low · eight hard validated tasks, effort ladder",
          "bContext": "Claude Code · eight hard validated tasks, effort ladder"
        },
        {
          "metric": "List-price cost per strict pass by effort (calculation)",
          "aValue": 0.02115,
          "bValue": 0.02893,
          "unit": "usd",
          "aDisplay": "$0.021",
          "bDisplay": "$0.029",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.021 vs $0.029) is not tested against run-to-run variation.",
          "studySlug": "effort-ladder",
          "n": 16,
          "chartId": "effort-ladder-cost-per-pass",
          "aN": 16,
          "bN": 16,
          "aContext": "Claude Code · effort low · eight hard validated tasks, effort ladder",
          "bContext": "Claude Code · eight hard validated tasks, effort ladder",
          "calculation": true
        },
        {
          "metric": "Reasoning cost per strict pass by effort, with the total (calculation) (Reasoning cost per strict pass)",
          "aValue": 0.005031,
          "bValue": 0.013104,
          "unit": "usd",
          "aDisplay": "$0.0050",
          "bDisplay": "$0.013",
          "winner": "unclear",
          "basis": "More or fewer usd is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "n": 16,
          "chartId": "thinking-bill-by-effort",
          "aN": 16,
          "bN": 16,
          "aContext": "Claude Code · effort low",
          "bContext": "Claude Code",
          "calculation": true
        },
        {
          "metric": "Reasoning cost per strict pass by effort, with the total (calculation) (Total cost per strict pass)",
          "aValue": 0.021152,
          "bValue": 0.028925,
          "unit": "usd",
          "aDisplay": "$0.021",
          "bDisplay": "$0.029",
          "winner": "unclear",
          "basis": "More or fewer usd is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "n": 16,
          "chartId": "thinking-bill-by-effort",
          "aN": 16,
          "bN": 16,
          "aContext": "Claude Code · effort low",
          "bContext": "Claude Code",
          "calculation": true
        }
      ]
    },
    {
      "slug": "claude-opus-5-5-medium-vs-high",
      "entity": "claude-opus-5-5",
      "aEffort": "medium",
      "bEffort": "high",
      "title": "Claude Opus 5.5: medium vs high effort",
      "seoTitle": "Claude Opus 5.5: medium vs high effort, measured",
      "description": "Claude Opus 5.5 at medium vs high effort: 3 measured metrics from one study, with sample sizes, intervals and every failure counted.",
      "verdict": "Claude Opus 5.5 at medium effort and Claude Opus 5.5 at high effort share 3 measured metrics and 3 list-price calculations from 2 studies. Only the effort setting differs between the two sides of a row; the route and the task set are the same. No row separates them: every interval or run range overlaps, too few runs were recorded, no interval was recorded, or more is not better for that metric. The rows are 1 tie and 5 unclear; each row says why. Calculation rows are derived from list prices and recorded counts; they are not bills or runs.",
      "rows": [
        {
          "metric": "Strict pass rate by effort on eight hard tasks",
          "aValue": 1,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "100% (16/16)",
          "bDisplay": "100% (16/16)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Opus 5.5 at medium effort 81% to 100%; Claude Opus 5.5 at high effort 81% to 100%), so this sample cannot separate them.",
          "studySlug": "effort-ladder",
          "n": 16,
          "chartId": "effort-ladder-pass-rate",
          "aN": 16,
          "bN": 16,
          "aContext": "Claude Code · effort medium · eight hard validated tasks, effort ladder",
          "bContext": "Claude Code · effort high · eight hard validated tasks, effort ladder",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.8064,
            1
          ],
          "bRange": [
            0.8064,
            1
          ]
        },
        {
          "metric": "Total time per call by effort on hard tasks",
          "aValue": 9.72,
          "bValue": 10.11,
          "unit": "seconds",
          "aDisplay": "9.72 s",
          "bDisplay": "10.1 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Opus 5.5 at medium effort 4.78 s to 31.4 s; Claude Opus 5.5 at high effort 3.63 s to 63.0 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "effort-ladder",
          "n": 16,
          "chartId": "effort-ladder-total-latency",
          "aN": 16,
          "bN": 16,
          "aContext": "Claude Code · effort medium · eight hard validated tasks, effort ladder",
          "bContext": "Claude Code · effort high · eight hard validated tasks, effort ladder",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            4.78,
            31.36
          ],
          "bRange": [
            3.63,
            63
          ]
        },
        {
          "metric": "Output tokens per call by effort on hard tasks (Output tokens)",
          "aValue": 853,
          "bValue": 1052,
          "unit": "tokens",
          "aDisplay": "853",
          "bDisplay": "1,052",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "effort-ladder",
          "n": 16,
          "chartId": "effort-ladder-output-tokens",
          "aN": 16,
          "bN": 16,
          "aContext": "Claude Code · effort medium · eight hard validated tasks, effort ladder",
          "bContext": "Claude Code · effort high · eight hard validated tasks, effort ladder"
        },
        {
          "metric": "List-price cost per strict pass by effort (calculation)",
          "aValue": 0.02947,
          "bValue": 0.03368,
          "unit": "usd",
          "aDisplay": "$0.029",
          "bDisplay": "$0.034",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.029 vs $0.034) is not tested against run-to-run variation.",
          "studySlug": "effort-ladder",
          "n": 16,
          "chartId": "effort-ladder-cost-per-pass",
          "aN": 16,
          "bN": 16,
          "aContext": "Claude Code · effort medium · eight hard validated tasks, effort ladder",
          "bContext": "Claude Code · effort high · eight hard validated tasks, effort ladder",
          "calculation": true
        },
        {
          "metric": "Reasoning cost per strict pass by effort, with the total (calculation) (Reasoning cost per strict pass)",
          "aValue": 0.01344,
          "bValue": 0.018034,
          "unit": "usd",
          "aDisplay": "$0.013",
          "bDisplay": "$0.018",
          "winner": "unclear",
          "basis": "More or fewer usd is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "n": 16,
          "chartId": "thinking-bill-by-effort",
          "aN": 16,
          "bN": 16,
          "aContext": "Claude Code · effort medium",
          "bContext": "Claude Code · effort high",
          "calculation": true
        },
        {
          "metric": "Reasoning cost per strict pass by effort, with the total (calculation) (Total cost per strict pass)",
          "aValue": 0.029475,
          "bValue": 0.033677,
          "unit": "usd",
          "aDisplay": "$0.029",
          "bDisplay": "$0.034",
          "winner": "unclear",
          "basis": "More or fewer usd is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "n": 16,
          "chartId": "thinking-bill-by-effort",
          "aN": 16,
          "bN": 16,
          "aContext": "Claude Code · effort medium",
          "bContext": "Claude Code · effort high",
          "calculation": true
        }
      ]
    },
    {
      "slug": "claude-opus-5-5-medium-vs-default",
      "entity": "claude-opus-5-5",
      "aEffort": "medium",
      "bEffort": "default",
      "title": "Claude Opus 5.5: medium vs default effort",
      "seoTitle": "Claude Opus 5.5: medium vs default effort, measured",
      "description": "Claude Opus 5.5 at medium vs default effort: 3 measured metrics from one study, with sample sizes, intervals and every failure counted.",
      "verdict": "Claude Opus 5.5 at medium effort and Claude Opus 5.5 at default effort share 3 measured metrics and 3 list-price calculations from 2 studies. Only the effort setting differs between the two sides of a row; the route and the task set are the same. \"default\" means the effort flag was not passed, so its level is the CLI’s choice. No row separates them: every interval or run range overlaps, too few runs were recorded, no interval was recorded, or more is not better for that metric. The rows are 1 tie and 5 unclear; each row says why. Calculation rows are derived from list prices and recorded counts; they are not bills or runs.",
      "rows": [
        {
          "metric": "Strict pass rate by effort on eight hard tasks",
          "aValue": 1,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "100% (16/16)",
          "bDisplay": "100% (16/16)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Opus 5.5 at medium effort 81% to 100%; Claude Opus 5.5 at default effort 81% to 100%), so this sample cannot separate them.",
          "studySlug": "effort-ladder",
          "n": 16,
          "chartId": "effort-ladder-pass-rate",
          "aN": 16,
          "bN": 16,
          "aContext": "Claude Code · effort medium · eight hard validated tasks, effort ladder",
          "bContext": "Claude Code · eight hard validated tasks, effort ladder",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.8064,
            1
          ],
          "bRange": [
            0.8064,
            1
          ]
        },
        {
          "metric": "Total time per call by effort on hard tasks",
          "aValue": 9.72,
          "bValue": 9.18,
          "unit": "seconds",
          "aDisplay": "9.72 s",
          "bDisplay": "9.18 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Opus 5.5 at medium effort 4.78 s to 31.4 s; Claude Opus 5.5 at default effort 4.24 s to 27.2 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "effort-ladder",
          "n": 16,
          "chartId": "effort-ladder-total-latency",
          "aN": 16,
          "bN": 16,
          "aContext": "Claude Code · effort medium · eight hard validated tasks, effort ladder",
          "bContext": "Claude Code · eight hard validated tasks, effort ladder",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            4.78,
            31.36
          ],
          "bRange": [
            4.24,
            27.21
          ]
        },
        {
          "metric": "Output tokens per call by effort on hard tasks (Output tokens)",
          "aValue": 853,
          "bValue": 945,
          "unit": "tokens",
          "aDisplay": "853",
          "bDisplay": "945",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "effort-ladder",
          "n": 16,
          "chartId": "effort-ladder-output-tokens",
          "aN": 16,
          "bN": 16,
          "aContext": "Claude Code · effort medium · eight hard validated tasks, effort ladder",
          "bContext": "Claude Code · eight hard validated tasks, effort ladder"
        },
        {
          "metric": "List-price cost per strict pass by effort (calculation)",
          "aValue": 0.02947,
          "bValue": 0.02893,
          "unit": "usd",
          "aDisplay": "$0.029",
          "bDisplay": "$0.029",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.029 vs $0.029) is not tested against run-to-run variation.",
          "studySlug": "effort-ladder",
          "n": 16,
          "chartId": "effort-ladder-cost-per-pass",
          "aN": 16,
          "bN": 16,
          "aContext": "Claude Code · effort medium · eight hard validated tasks, effort ladder",
          "bContext": "Claude Code · eight hard validated tasks, effort ladder",
          "calculation": true
        },
        {
          "metric": "Reasoning cost per strict pass by effort, with the total (calculation) (Reasoning cost per strict pass)",
          "aValue": 0.01344,
          "bValue": 0.013104,
          "unit": "usd",
          "aDisplay": "$0.013",
          "bDisplay": "$0.013",
          "winner": "unclear",
          "basis": "More or fewer usd is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "n": 16,
          "chartId": "thinking-bill-by-effort",
          "aN": 16,
          "bN": 16,
          "aContext": "Claude Code · effort medium",
          "bContext": "Claude Code",
          "calculation": true
        },
        {
          "metric": "Reasoning cost per strict pass by effort, with the total (calculation) (Total cost per strict pass)",
          "aValue": 0.029475,
          "bValue": 0.028925,
          "unit": "usd",
          "aDisplay": "$0.029",
          "bDisplay": "$0.029",
          "winner": "unclear",
          "basis": "More or fewer usd is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "n": 16,
          "chartId": "thinking-bill-by-effort",
          "aN": 16,
          "bN": 16,
          "aContext": "Claude Code · effort medium",
          "bContext": "Claude Code",
          "calculation": true
        }
      ]
    },
    {
      "slug": "claude-opus-5-5-high-vs-default",
      "entity": "claude-opus-5-5",
      "aEffort": "high",
      "bEffort": "default",
      "title": "Claude Opus 5.5: high vs default effort",
      "seoTitle": "Claude Opus 5.5: high vs default effort, measured",
      "description": "Claude Opus 5.5 at high vs default effort: 14 measured metrics from 3 studies, with sample sizes, intervals and every failure counted.",
      "verdict": "Claude Opus 5.5 at high effort and Claude Opus 5.5 at default effort share 14 measured metrics and 12 list-price calculations from 4 studies. Only the effort setting differs between the two sides of a row; the route and the task set are the same. \"default\" means the effort flag was not passed, so its level is the CLI’s choice. No row separates them: every interval or run range overlaps, too few runs were recorded, no interval was recorded, or more is not better for that metric. The rows are 4 ties and 22 unclear; each row says why. Calculation rows are derived from list prices and recorded counts; they are not bills or runs. Some rows rest on small samples (n = 15 at the smallest).",
      "rows": [
        {
          "metric": "Pass rate on five validated tasks",
          "aValue": 1,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "100% (15/15)",
          "bDisplay": "100% (15/15)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Opus 5.5 at high effort 80% to 100%; Claude Opus 5.5 at default effort 80% to 100%), so this sample cannot separate them.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-pass-rate",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code · effort high · five short validated tasks",
          "bContext": "Claude Code · five short validated tasks",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.7961,
            1
          ],
          "bRange": [
            0.7961,
            1
          ]
        },
        {
          "metric": "Total time per call",
          "aValue": 2.71,
          "bValue": 2.75,
          "unit": "seconds",
          "aDisplay": "2.71 s",
          "bDisplay": "2.75 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Opus 5.5 at high effort 2.45 s to 11.8 s; Claude Opus 5.5 at default effort 2.47 s to 8.91 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-total-latency",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code · effort high · five short validated tasks",
          "bContext": "Claude Code · five short validated tasks",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            2.45,
            11.78
          ],
          "bRange": [
            2.47,
            8.91
          ]
        },
        {
          "metric": "Time to first useful output",
          "aValue": 2.04,
          "bValue": 1.92,
          "unit": "seconds",
          "aDisplay": "2.04 s",
          "bDisplay": "1.92 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Opus 5.5 at high effort 1.40 s to 9.94 s; Claude Opus 5.5 at default effort 1.56 s to 7.23 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-first-useful-latency",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code · effort high · five short validated tasks",
          "bContext": "Claude Code · five short validated tasks",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            1.4,
            9.94
          ],
          "bRange": [
            1.56,
            7.23
          ]
        },
        {
          "metric": "Input tokens per call: what the CLI sends (Cache read)",
          "aValue": 1463,
          "bValue": 1401,
          "unit": "tokens",
          "aDisplay": "1,463",
          "bDisplay": "1,401",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-input-tokens",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code · effort high · five short validated tasks",
          "bContext": "Claude Code · five short validated tasks"
        },
        {
          "metric": "Input tokens per call: what the CLI sends (Other input)",
          "aValue": 619,
          "bValue": 680,
          "unit": "tokens",
          "aDisplay": "619",
          "bDisplay": "680",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-input-tokens",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code · effort high · five short validated tasks",
          "bContext": "Claude Code · five short validated tasks"
        },
        {
          "metric": "Output tokens per call (Output tokens)",
          "aValue": 78,
          "bValue": 64,
          "unit": "tokens",
          "aDisplay": "78",
          "bDisplay": "64",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-output-tokens",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code · effort high · five short validated tasks",
          "bContext": "Claude Code · five short validated tasks"
        },
        {
          "metric": "List-price cost per call (calculation)",
          "aValue": 0.00694,
          "bValue": 0.00688,
          "unit": "usd",
          "aDisplay": "$0.0069",
          "bDisplay": "$0.0069",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Opus 5.5 at high effort $0.0059 to $0.027; Claude Opus 5.5 at default effort $0.0059 to $0.022); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-list-price-per-call",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code · effort high · five short validated tasks",
          "bContext": "Claude Code · five short validated tasks",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            0.00592,
            0.02708
          ],
          "bRange": [
            0.00592,
            0.02226
          ],
          "calculation": true
        },
        {
          "metric": "List-price cost per passing answer (calculation)",
          "aValue": 0.01049,
          "bValue": 0.01009,
          "unit": "usd",
          "aDisplay": "$0.010",
          "bDisplay": "$0.010",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.010 vs $0.010) is not tested against run-to-run variation.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-cost-per-pass",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code · effort high · five short validated tasks",
          "bContext": "Claude Code · five short validated tasks",
          "calculation": true
        },
        {
          "metric": "Pass rate on eight hard tasks (Strict pass)",
          "aValue": 1,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "100% (24/24)",
          "bDisplay": "100% (24/24)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Opus 5.5 at high effort 86% to 100%; Claude Opus 5.5 at default effort 86% to 100%), so this sample cannot separate them.",
          "studySlug": "hard-model-head-to-head",
          "n": 24,
          "chartId": "hard-h2h-pass-rate",
          "aN": 24,
          "bN": 24,
          "aContext": "Claude Code · effort high · eight hard validated tasks",
          "bContext": "Claude Code · eight hard validated tasks",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.862,
            1
          ],
          "bRange": [
            0.862,
            1
          ]
        },
        {
          "metric": "Pass rate on eight hard tasks (Lenient (format misses counted))",
          "aValue": 1,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "100% (24/24)",
          "bDisplay": "100% (24/24)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Opus 5.5 at high effort 86% to 100%; Claude Opus 5.5 at default effort 86% to 100%), so this sample cannot separate them.",
          "studySlug": "hard-model-head-to-head",
          "n": 24,
          "chartId": "hard-h2h-pass-rate",
          "aN": 24,
          "bN": 24,
          "aContext": "Claude Code · effort high · eight hard validated tasks",
          "bContext": "Claude Code · eight hard validated tasks",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.862,
            1
          ],
          "bRange": [
            0.862,
            1
          ]
        },
        {
          "metric": "Total time per call on hard tasks (separate batches)",
          "aValue": 11.03,
          "bValue": 9.18,
          "unit": "seconds",
          "aDisplay": "11.0 s",
          "bDisplay": "9.18 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Opus 5.5 at high effort 3.63 s to 63.0 s; Claude Opus 5.5 at default effort 4.24 s to 27.2 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "hard-model-head-to-head",
          "n": 24,
          "chartId": "hard-h2h-total-latency",
          "aN": 24,
          "bN": 24,
          "aContext": "Claude Code · effort high · eight hard validated tasks",
          "bContext": "Claude Code · eight hard validated tasks",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            3.63,
            63
          ],
          "bRange": [
            4.24,
            27.21
          ]
        },
        {
          "metric": "Time to first useful output on hard tasks",
          "aValue": 7.13,
          "bValue": 6.78,
          "unit": "seconds",
          "aDisplay": "7.13 s",
          "bDisplay": "6.78 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Opus 5.5 at high effort 2.15 s to 56.2 s; Claude Opus 5.5 at default effort 2.39 s to 21.8 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "hard-model-head-to-head",
          "n": 24,
          "chartId": "hard-h2h-first-useful-latency",
          "aN": 24,
          "bN": 24,
          "aContext": "Claude Code · effort high · eight hard validated tasks",
          "bContext": "Claude Code · eight hard validated tasks",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            2.15,
            56.23
          ],
          "bRange": [
            2.39,
            21.77
          ]
        },
        {
          "metric": "Output tokens per call on hard tasks (Output tokens)",
          "aValue": 1052,
          "bValue": 945,
          "unit": "tokens",
          "aDisplay": "1,052",
          "bDisplay": "945",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "hard-model-head-to-head",
          "n": 24,
          "chartId": "hard-h2h-output-tokens",
          "aN": 24,
          "bN": 24,
          "aContext": "Claude Code · effort high · eight hard validated tasks",
          "bContext": "Claude Code · eight hard validated tasks"
        },
        {
          "metric": "List-price cost per strict pass on hard tasks (calculation)",
          "aValue": 0.03337,
          "bValue": 0.02824,
          "unit": "usd",
          "aDisplay": "$0.033",
          "bDisplay": "$0.028",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.033 vs $0.028) is not tested against run-to-run variation.",
          "studySlug": "hard-model-head-to-head",
          "n": 24,
          "chartId": "hard-h2h-cost-per-pass",
          "aN": 24,
          "bN": 24,
          "aContext": "Claude Code · effort high · eight hard validated tasks",
          "bContext": "Claude Code · eight hard validated tasks",
          "calculation": true
        },
        {
          "metric": "Strict pass rate by effort on eight hard tasks",
          "aValue": 1,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "100% (16/16)",
          "bDisplay": "100% (16/16)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (Claude Opus 5.5 at high effort 81% to 100%; Claude Opus 5.5 at default effort 81% to 100%), so this sample cannot separate them.",
          "studySlug": "effort-ladder",
          "n": 16,
          "chartId": "effort-ladder-pass-rate",
          "aN": 16,
          "bN": 16,
          "aContext": "Claude Code · effort high · eight hard validated tasks, effort ladder",
          "bContext": "Claude Code · eight hard validated tasks, effort ladder",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.8064,
            1
          ],
          "bRange": [
            0.8064,
            1
          ]
        },
        {
          "metric": "Total time per call by effort on hard tasks",
          "aValue": 10.11,
          "bValue": 9.18,
          "unit": "seconds",
          "aDisplay": "10.1 s",
          "bDisplay": "9.18 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (Claude Opus 5.5 at high effort 3.63 s to 63.0 s; Claude Opus 5.5 at default effort 4.24 s to 27.2 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "effort-ladder",
          "n": 16,
          "chartId": "effort-ladder-total-latency",
          "aN": 16,
          "bN": 16,
          "aContext": "Claude Code · effort high · eight hard validated tasks, effort ladder",
          "bContext": "Claude Code · eight hard validated tasks, effort ladder",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            3.63,
            63
          ],
          "bRange": [
            4.24,
            27.21
          ]
        },
        {
          "metric": "Output tokens per call by effort on hard tasks (Output tokens)",
          "aValue": 1052,
          "bValue": 945,
          "unit": "tokens",
          "aDisplay": "1,052",
          "bDisplay": "945",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "effort-ladder",
          "n": 16,
          "chartId": "effort-ladder-output-tokens",
          "aN": 16,
          "bN": 16,
          "aContext": "Claude Code · effort high · eight hard validated tasks, effort ladder",
          "bContext": "Claude Code · eight hard validated tasks, effort ladder"
        },
        {
          "metric": "List-price cost per strict pass by effort (calculation)",
          "aValue": 0.03368,
          "bValue": 0.02893,
          "unit": "usd",
          "aDisplay": "$0.034",
          "bDisplay": "$0.029",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.034 vs $0.029) is not tested against run-to-run variation.",
          "studySlug": "effort-ladder",
          "n": 16,
          "chartId": "effort-ladder-cost-per-pass",
          "aN": 16,
          "bN": 16,
          "aContext": "Claude Code · effort high · eight hard validated tasks, effort ladder",
          "bContext": "Claude Code · eight hard validated tasks, effort ladder",
          "calculation": true
        },
        {
          "metric": "Reasoning share of output tokens per call on hard tasks (calculation)",
          "aValue": 54.43,
          "bValue": 54.79,
          "unit": "percent",
          "aDisplay": "54.4%",
          "bDisplay": "54.8%",
          "winner": "unclear",
          "basis": "More or fewer percent is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "n": 24,
          "chartId": "thinking-bill-share",
          "aN": 24,
          "bN": 24,
          "aContext": "Claude Code · effort high",
          "bContext": "Claude Code",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            36.14,
            96.23
          ],
          "bRange": [
            29.92,
            95.6
          ],
          "calculation": true
        },
        {
          "metric": "List-price cost per call: reasoning, remaining output and input (calculation) (Reasoning (output tokens))",
          "aValue": 0.017969,
          "bValue": 0.012528,
          "unit": "usd",
          "aDisplay": "$0.018",
          "bDisplay": "$0.013",
          "winner": "unclear",
          "basis": "More or fewer usd is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "n": 24,
          "chartId": "thinking-bill-cost-per-call",
          "aN": 24,
          "bN": 24,
          "aContext": "Claude Code · effort high",
          "bContext": "Claude Code",
          "calculation": true
        },
        {
          "metric": "List-price cost per call: reasoning, remaining output and input (calculation) (Remaining output (visible-answer estimate))",
          "aValue": 0.00799,
          "bValue": 0.008003,
          "unit": "usd",
          "aDisplay": "$0.0080",
          "bDisplay": "$0.0080",
          "winner": "unclear",
          "basis": "More or fewer usd is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "n": 24,
          "chartId": "thinking-bill-cost-per-call",
          "aN": 24,
          "bN": 24,
          "aContext": "Claude Code · effort high",
          "bContext": "Claude Code",
          "calculation": true
        },
        {
          "metric": "List-price cost per call: reasoning, remaining output and input (calculation) (Input (prompt, cache priced))",
          "aValue": 0.007407,
          "bValue": 0.00771,
          "unit": "usd",
          "aDisplay": "$0.0074",
          "bDisplay": "$0.0077",
          "winner": "unclear",
          "basis": "More or fewer usd is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "n": 24,
          "chartId": "thinking-bill-cost-per-call",
          "aN": 24,
          "bN": 24,
          "aContext": "Claude Code · effort high",
          "bContext": "Claude Code",
          "calculation": true
        },
        {
          "metric": "Reasoning cost per strict pass by effort, with the total (calculation) (Reasoning cost per strict pass)",
          "aValue": 0.018034,
          "bValue": 0.013104,
          "unit": "usd",
          "aDisplay": "$0.018",
          "bDisplay": "$0.013",
          "winner": "unclear",
          "basis": "More or fewer usd is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "n": 16,
          "chartId": "thinking-bill-by-effort",
          "aN": 16,
          "bN": 16,
          "aContext": "Claude Code · effort high",
          "bContext": "Claude Code",
          "calculation": true
        },
        {
          "metric": "Reasoning cost per strict pass by effort, with the total (calculation) (Total cost per strict pass)",
          "aValue": 0.033677,
          "bValue": 0.028925,
          "unit": "usd",
          "aDisplay": "$0.034",
          "bDisplay": "$0.029",
          "winner": "unclear",
          "basis": "More or fewer usd is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "n": 16,
          "chartId": "thinking-bill-by-effort",
          "aN": 16,
          "bN": 16,
          "aContext": "Claude Code · effort high",
          "bContext": "Claude Code",
          "calculation": true
        },
        {
          "metric": "Reasoning share on short tasks vs hard tasks (calculation) (Eight hard tasks)",
          "aValue": 54.43,
          "bValue": 54.79,
          "unit": "percent",
          "aDisplay": "54.4%",
          "bDisplay": "54.8%",
          "winner": "unclear",
          "basis": "More or fewer percent is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "n": 24,
          "chartId": "thinking-bill-short-vs-hard",
          "aN": 24,
          "bN": 24,
          "aContext": "Claude Code · effort high",
          "bContext": "Claude Code",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            36.14,
            96.23
          ],
          "bRange": [
            29.92,
            95.6
          ],
          "calculation": true
        },
        {
          "metric": "Reasoning share on short tasks vs hard tasks (calculation) (Five short tasks)",
          "aValue": 43.59,
          "bValue": 0,
          "unit": "percent",
          "aDisplay": "43.6%",
          "bDisplay": "0%",
          "winner": "unclear",
          "basis": "More or fewer percent is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "n": 15,
          "chartId": "thinking-bill-short-vs-hard",
          "aN": 15,
          "bN": 15,
          "aContext": "Claude Code · effort high",
          "bContext": "Claude Code",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            0,
            93.33
          ],
          "bRange": [
            0,
            93.33
          ],
          "calculation": true
        }
      ]
    },
    {
      "slug": "gpt-6-1-sol-codex-cli-low-vs-medium",
      "entity": "gpt-6-1-sol-codex-cli",
      "aEffort": "low",
      "bEffort": "medium",
      "title": "GPT-6.1 Sol (Codex CLI): low vs medium effort",
      "seoTitle": "GPT-6.1 Sol (Codex CLI): low vs medium effort, measured",
      "description": "GPT-6.1 Sol (Codex CLI) at low vs medium effort: 9 measured metrics from 2 studies, with sample sizes, intervals and every failure counted.",
      "verdict": "GPT-6.1 Sol (Codex CLI) at low effort and GPT-6.1 Sol (Codex CLI) at medium effort share 9 measured metrics and 5 list-price calculations from 3 studies. Only the effort setting differs between the two sides of a row; the route and the task set are the same. No row separates them: every interval or run range overlaps, too few runs were recorded, no interval was recorded, or more is not better for that metric. The rows are 3 ties and 11 unclear; each row says why. Calculation rows are derived from list prices and recorded counts; they are not bills or runs. Some rows rest on small samples (n = 10 at the smallest).",
      "rows": [
        {
          "metric": "Pass rate on five validated tasks",
          "aValue": 1,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "100% (10/10)",
          "bDisplay": "100% (15/15)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (GPT-6.1 Sol (Codex CLI) at low effort 72% to 100%; GPT-6.1 Sol (Codex CLI) at medium effort 80% to 100%), so this sample cannot separate them.",
          "studySlug": "model-head-to-head",
          "chartId": "h2h-pass-rate",
          "aN": 10,
          "bN": 15,
          "aContext": "Codex CLI · effort low · five short validated tasks",
          "bContext": "Codex CLI · effort medium · five short validated tasks",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.7225,
            1
          ],
          "bRange": [
            0.7961,
            1
          ]
        },
        {
          "metric": "Total time per call",
          "aValue": 6.26,
          "bValue": 5.65,
          "unit": "seconds",
          "aDisplay": "6.26 s",
          "bDisplay": "5.65 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (GPT-6.1 Sol (Codex CLI) at low effort 4.65 s to 10.5 s; GPT-6.1 Sol (Codex CLI) at medium effort 4.10 s to 25.5 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "model-head-to-head",
          "chartId": "h2h-total-latency",
          "aN": 10,
          "bN": 15,
          "aContext": "Codex CLI · effort low · five short validated tasks",
          "bContext": "Codex CLI · effort medium · five short validated tasks",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            4.65,
            10.47
          ],
          "bRange": [
            4.1,
            25.46
          ]
        },
        {
          "metric": "Time to first useful output",
          "aValue": 5.14,
          "bValue": 5.05,
          "unit": "seconds",
          "aDisplay": "5.14 s",
          "bDisplay": "5.05 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (GPT-6.1 Sol (Codex CLI) at low effort 4.02 s to 8.50 s; GPT-6.1 Sol (Codex CLI) at medium effort 3.36 s to 17.8 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "model-head-to-head",
          "chartId": "h2h-first-useful-latency",
          "aN": 10,
          "bN": 15,
          "aContext": "Codex CLI · effort low · five short validated tasks",
          "bContext": "Codex CLI · effort medium · five short validated tasks",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            4.02,
            8.5
          ],
          "bRange": [
            3.36,
            17.82
          ]
        },
        {
          "metric": "Input tokens per call: what the CLI sends (Cache read)",
          "aValue": 8064,
          "bValue": 5180,
          "unit": "tokens",
          "aDisplay": "8,064",
          "bDisplay": "5,180",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "model-head-to-head",
          "chartId": "h2h-input-tokens",
          "aN": 10,
          "bN": 15,
          "aContext": "Codex CLI · effort low · five short validated tasks",
          "bContext": "Codex CLI · effort medium · five short validated tasks"
        },
        {
          "metric": "Input tokens per call: what the CLI sends (Other input)",
          "aValue": 4059,
          "bValue": 6943,
          "unit": "tokens",
          "aDisplay": "4,059",
          "bDisplay": "6,943",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "model-head-to-head",
          "chartId": "h2h-input-tokens",
          "aN": 10,
          "bN": 15,
          "aContext": "Codex CLI · effort low · five short validated tasks",
          "bContext": "Codex CLI · effort medium · five short validated tasks"
        },
        {
          "metric": "Output tokens per call (Output tokens)",
          "aValue": 42,
          "bValue": 42,
          "unit": "tokens",
          "aDisplay": "42",
          "bDisplay": "42",
          "winner": "tie",
          "basis": "Same value. More or fewer is not better by itself for this metric.",
          "studySlug": "model-head-to-head",
          "chartId": "h2h-output-tokens",
          "aN": 10,
          "bN": 15,
          "aContext": "Codex CLI · effort low · five short validated tasks",
          "bContext": "Codex CLI · effort medium · five short validated tasks"
        },
        {
          "metric": "List-price cost per call (calculation)",
          "aValue": 0.00769,
          "bValue": 0.01018,
          "unit": "usd",
          "aDisplay": "$0.0077",
          "bDisplay": "$0.010",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (GPT-6.1 Sol (Codex CLI) at low effort $0.0074 to $0.026; GPT-6.1 Sol (Codex CLI) at medium effort $0.0054 to $0.027); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "model-head-to-head",
          "chartId": "h2h-list-price-per-call",
          "aN": 10,
          "bN": 15,
          "aContext": "Codex CLI · effort low · five short validated tasks",
          "bContext": "Codex CLI · effort medium · five short validated tasks",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            0.00742,
            0.02649
          ],
          "bRange": [
            0.0054,
            0.02686
          ],
          "calculation": true
        },
        {
          "metric": "List-price cost per passing answer (calculation)",
          "aValue": 0.00998,
          "bValue": 0.01564,
          "unit": "usd",
          "aDisplay": "$0.010",
          "bDisplay": "$0.016",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.010 vs $0.016) is not tested against run-to-run variation.",
          "studySlug": "model-head-to-head",
          "chartId": "h2h-cost-per-pass",
          "aN": 10,
          "bN": 15,
          "aContext": "Codex CLI · effort low · five short validated tasks",
          "bContext": "Codex CLI · effort medium · five short validated tasks",
          "calculation": true
        },
        {
          "metric": "Strict pass rate by effort on eight hard tasks",
          "aValue": 1,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "100% (16/16)",
          "bDisplay": "100% (16/16)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (GPT-6.1 Sol (Codex CLI) at low effort 81% to 100%; GPT-6.1 Sol (Codex CLI) at medium effort 81% to 100%), so this sample cannot separate them.",
          "studySlug": "effort-ladder",
          "n": 16,
          "chartId": "effort-ladder-pass-rate",
          "aN": 16,
          "bN": 16,
          "aContext": "Codex CLI · effort low · eight hard validated tasks, effort ladder",
          "bContext": "Codex CLI · effort medium · eight hard validated tasks, effort ladder",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.8064,
            1
          ],
          "bRange": [
            0.8064,
            1
          ]
        },
        {
          "metric": "Total time per call by effort on hard tasks",
          "aValue": 13.62,
          "bValue": 13.11,
          "unit": "seconds",
          "aDisplay": "13.6 s",
          "bDisplay": "13.1 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (GPT-6.1 Sol (Codex CLI) at low effort 7.94 s to 44.3 s; GPT-6.1 Sol (Codex CLI) at medium effort 8.54 s to 61.6 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "effort-ladder",
          "n": 16,
          "chartId": "effort-ladder-total-latency",
          "aN": 16,
          "bN": 16,
          "aContext": "Codex CLI · effort low · eight hard validated tasks, effort ladder",
          "bContext": "Codex CLI · effort medium · eight hard validated tasks, effort ladder",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            7.94,
            44.29
          ],
          "bRange": [
            8.54,
            61.6
          ]
        },
        {
          "metric": "Output tokens per call by effort on hard tasks (Output tokens)",
          "aValue": 284,
          "bValue": 335,
          "unit": "tokens",
          "aDisplay": "284",
          "bDisplay": "335",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "effort-ladder",
          "n": 16,
          "chartId": "effort-ladder-output-tokens",
          "aN": 16,
          "bN": 16,
          "aContext": "Codex CLI · effort low · eight hard validated tasks, effort ladder",
          "bContext": "Codex CLI · effort medium · eight hard validated tasks, effort ladder"
        },
        {
          "metric": "List-price cost per strict pass by effort (calculation)",
          "aValue": 0.01284,
          "bValue": 0.02564,
          "unit": "usd",
          "aDisplay": "$0.013",
          "bDisplay": "$0.026",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.013 vs $0.026) is not tested against run-to-run variation.",
          "studySlug": "effort-ladder",
          "n": 16,
          "chartId": "effort-ladder-cost-per-pass",
          "aN": 16,
          "bN": 16,
          "aContext": "Codex CLI · effort low · eight hard validated tasks, effort ladder",
          "bContext": "Codex CLI · effort medium · eight hard validated tasks, effort ladder",
          "calculation": true
        },
        {
          "metric": "Reasoning cost per strict pass by effort, with the total (calculation) (Reasoning cost per strict pass)",
          "aValue": 0.001223,
          "bValue": 0.002273,
          "unit": "usd",
          "aDisplay": "$0.0012",
          "bDisplay": "$0.0023",
          "winner": "unclear",
          "basis": "More or fewer usd is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "n": 16,
          "chartId": "thinking-bill-by-effort",
          "aN": 16,
          "bN": 16,
          "aContext": "Codex CLI · effort low",
          "bContext": "Codex CLI · effort medium",
          "calculation": true
        },
        {
          "metric": "Reasoning cost per strict pass by effort, with the total (calculation) (Total cost per strict pass)",
          "aValue": 0.012837,
          "bValue": 0.025637,
          "unit": "usd",
          "aDisplay": "$0.013",
          "bDisplay": "$0.026",
          "winner": "unclear",
          "basis": "More or fewer usd is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "n": 16,
          "chartId": "thinking-bill-by-effort",
          "aN": 16,
          "bN": 16,
          "aContext": "Codex CLI · effort low",
          "bContext": "Codex CLI · effort medium",
          "calculation": true
        }
      ]
    },
    {
      "slug": "gpt-6-1-sol-codex-cli-low-vs-high",
      "entity": "gpt-6-1-sol-codex-cli",
      "aEffort": "low",
      "bEffort": "high",
      "title": "GPT-6.1 Sol (Codex CLI): low vs high effort",
      "seoTitle": "GPT-6.1 Sol (Codex CLI): low vs high effort, measured",
      "description": "GPT-6.1 Sol (Codex CLI) at low vs high effort: 14 measured metrics from 3 studies, with sample sizes, intervals and every failure counted.",
      "verdict": "GPT-6.1 Sol (Codex CLI) at low effort and GPT-6.1 Sol (Codex CLI) at high effort share 14 measured metrics and 5 list-price calculations from 4 studies. Only the effort setting differs between the two sides of a row; the route and the task set are the same. No row separates them: every interval or run range overlaps, too few runs were recorded, no interval was recorded, or more is not better for that metric. The rows are 3 ties and 16 unclear; each row says why. Calculation rows are derived from list prices and recorded counts; they are not bills or runs. Some rows rest on small samples (n = 3 at the smallest).",
      "rows": [
        {
          "metric": "Pass rate on five validated tasks",
          "aValue": 1,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "100% (10/10)",
          "bDisplay": "100% (15/15)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (GPT-6.1 Sol (Codex CLI) at low effort 72% to 100%; GPT-6.1 Sol (Codex CLI) at high effort 80% to 100%), so this sample cannot separate them.",
          "studySlug": "model-head-to-head",
          "chartId": "h2h-pass-rate",
          "aN": 10,
          "bN": 15,
          "aContext": "Codex CLI · effort low · five short validated tasks",
          "bContext": "Codex CLI · effort high · five short validated tasks",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.7225,
            1
          ],
          "bRange": [
            0.7961,
            1
          ]
        },
        {
          "metric": "Total time per call",
          "aValue": 6.26,
          "bValue": 5.6,
          "unit": "seconds",
          "aDisplay": "6.26 s",
          "bDisplay": "5.60 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (GPT-6.1 Sol (Codex CLI) at low effort 4.65 s to 10.5 s; GPT-6.1 Sol (Codex CLI) at high effort 4.05 s to 19.5 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "model-head-to-head",
          "chartId": "h2h-total-latency",
          "aN": 10,
          "bN": 15,
          "aContext": "Codex CLI · effort low · five short validated tasks",
          "bContext": "Codex CLI · effort high · five short validated tasks",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            4.65,
            10.47
          ],
          "bRange": [
            4.05,
            19.52
          ]
        },
        {
          "metric": "Time to first useful output",
          "aValue": 5.14,
          "bValue": 5.32,
          "unit": "seconds",
          "aDisplay": "5.14 s",
          "bDisplay": "5.32 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (GPT-6.1 Sol (Codex CLI) at low effort 4.02 s to 8.50 s; GPT-6.1 Sol (Codex CLI) at high effort 3.64 s to 16.4 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "model-head-to-head",
          "chartId": "h2h-first-useful-latency",
          "aN": 10,
          "bN": 15,
          "aContext": "Codex CLI · effort low · five short validated tasks",
          "bContext": "Codex CLI · effort high · five short validated tasks",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            4.02,
            8.5
          ],
          "bRange": [
            3.64,
            16.37
          ]
        },
        {
          "metric": "Input tokens per call: what the CLI sends (Cache read)",
          "aValue": 8064,
          "bValue": 6716,
          "unit": "tokens",
          "aDisplay": "8,064",
          "bDisplay": "6,716",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "model-head-to-head",
          "chartId": "h2h-input-tokens",
          "aN": 10,
          "bN": 15,
          "aContext": "Codex CLI · effort low · five short validated tasks",
          "bContext": "Codex CLI · effort high · five short validated tasks"
        },
        {
          "metric": "Input tokens per call: what the CLI sends (Other input)",
          "aValue": 4059,
          "bValue": 5406,
          "unit": "tokens",
          "aDisplay": "4,059",
          "bDisplay": "5,406",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "model-head-to-head",
          "chartId": "h2h-input-tokens",
          "aN": 10,
          "bN": 15,
          "aContext": "Codex CLI · effort low · five short validated tasks",
          "bContext": "Codex CLI · effort high · five short validated tasks"
        },
        {
          "metric": "Output tokens per call (Output tokens)",
          "aValue": 42,
          "bValue": 42,
          "unit": "tokens",
          "aDisplay": "42",
          "bDisplay": "42",
          "winner": "tie",
          "basis": "Same value. More or fewer is not better by itself for this metric.",
          "studySlug": "model-head-to-head",
          "chartId": "h2h-output-tokens",
          "aN": 10,
          "bN": 15,
          "aContext": "Codex CLI · effort low · five short validated tasks",
          "bContext": "Codex CLI · effort high · five short validated tasks"
        },
        {
          "metric": "List-price cost per call (calculation)",
          "aValue": 0.00769,
          "bValue": 0.01047,
          "unit": "usd",
          "aDisplay": "$0.0077",
          "bDisplay": "$0.010",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (GPT-6.1 Sol (Codex CLI) at low effort $0.0074 to $0.026; GPT-6.1 Sol (Codex CLI) at high effort $0.0066 to $0.028); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "model-head-to-head",
          "chartId": "h2h-list-price-per-call",
          "aN": 10,
          "bN": 15,
          "aContext": "Codex CLI · effort low · five short validated tasks",
          "bContext": "Codex CLI · effort high · five short validated tasks",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            0.00742,
            0.02649
          ],
          "bRange": [
            0.0066,
            0.02812
          ],
          "calculation": true
        },
        {
          "metric": "List-price cost per passing answer (calculation)",
          "aValue": 0.00998,
          "bValue": 0.01322,
          "unit": "usd",
          "aDisplay": "$0.010",
          "bDisplay": "$0.013",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.010 vs $0.013) is not tested against run-to-run variation.",
          "studySlug": "model-head-to-head",
          "chartId": "h2h-cost-per-pass",
          "aN": 10,
          "bN": 15,
          "aContext": "Codex CLI · effort low · five short validated tasks",
          "bContext": "Codex CLI · effort high · five short validated tasks",
          "calculation": true
        },
        {
          "metric": "Strict pass rate by effort on eight hard tasks",
          "aValue": 1,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "100% (16/16)",
          "bDisplay": "100% (16/16)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (GPT-6.1 Sol (Codex CLI) at low effort 81% to 100%; GPT-6.1 Sol (Codex CLI) at high effort 81% to 100%), so this sample cannot separate them.",
          "studySlug": "effort-ladder",
          "n": 16,
          "chartId": "effort-ladder-pass-rate",
          "aN": 16,
          "bN": 16,
          "aContext": "Codex CLI · effort low · eight hard validated tasks, effort ladder",
          "bContext": "Codex CLI · effort high · eight hard validated tasks, effort ladder",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.8064,
            1
          ],
          "bRange": [
            0.8064,
            1
          ]
        },
        {
          "metric": "Total time per call by effort on hard tasks",
          "aValue": 13.62,
          "bValue": 18.12,
          "unit": "seconds",
          "aDisplay": "13.6 s",
          "bDisplay": "18.1 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (GPT-6.1 Sol (Codex CLI) at low effort 7.94 s to 44.3 s; GPT-6.1 Sol (Codex CLI) at high effort 11.7 s to 92.2 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "effort-ladder",
          "n": 16,
          "chartId": "effort-ladder-total-latency",
          "aN": 16,
          "bN": 16,
          "aContext": "Codex CLI · effort low · eight hard validated tasks, effort ladder",
          "bContext": "Codex CLI · effort high · eight hard validated tasks, effort ladder",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            7.94,
            44.29
          ],
          "bRange": [
            11.67,
            92.21
          ]
        },
        {
          "metric": "Output tokens per call by effort on hard tasks (Output tokens)",
          "aValue": 284,
          "bValue": 436,
          "unit": "tokens",
          "aDisplay": "284",
          "bDisplay": "436",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "effort-ladder",
          "n": 16,
          "chartId": "effort-ladder-output-tokens",
          "aN": 16,
          "bN": 16,
          "aContext": "Codex CLI · effort low · eight hard validated tasks, effort ladder",
          "bContext": "Codex CLI · effort high · eight hard validated tasks, effort ladder"
        },
        {
          "metric": "List-price cost per strict pass by effort (calculation)",
          "aValue": 0.01284,
          "bValue": 0.01514,
          "unit": "usd",
          "aDisplay": "$0.013",
          "bDisplay": "$0.015",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.013 vs $0.015) is not tested against run-to-run variation.",
          "studySlug": "effort-ladder",
          "n": 16,
          "chartId": "effort-ladder-cost-per-pass",
          "aN": 16,
          "bN": 16,
          "aContext": "Codex CLI · effort low · eight hard validated tasks, effort ladder",
          "bContext": "Codex CLI · effort high · eight hard validated tasks, effort ladder",
          "calculation": true
        },
        {
          "metric": "CLI vs API: time for a one-line answer (Total time)",
          "aValue": 4.18,
          "bValue": 4.19,
          "unit": "seconds",
          "aDisplay": "4.18 s",
          "bDisplay": "4.19 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (GPT-6.1 Sol (Codex CLI) at low effort 3.86 s to 4.53 s; GPT-6.1 Sol (Codex CLI) at high effort 3.81 s to 4.69 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "cli-model-latency-tokens",
          "n": 5,
          "chartId": "cli-vs-api-exact-reply-latency",
          "aN": 5,
          "bN": 5,
          "aContext": "Codex CLI · effort low · fixed exact reply, 5 runs",
          "bContext": "Codex CLI · effort high · fixed exact reply, 5 runs",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            3.86,
            4.53
          ],
          "bRange": [
            3.81,
            4.69
          ]
        },
        {
          "metric": "CLI vs API: time for a one-line answer (First useful output)",
          "aValue": 3.75,
          "bValue": 3.79,
          "unit": "seconds",
          "aDisplay": "3.75 s",
          "bDisplay": "3.79 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (GPT-6.1 Sol (Codex CLI) at low effort 3.44 s to 4.10 s; GPT-6.1 Sol (Codex CLI) at high effort 3.37 s to 4.30 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "cli-model-latency-tokens",
          "n": 5,
          "chartId": "cli-vs-api-exact-reply-latency",
          "aN": 5,
          "bN": 5,
          "aContext": "Codex CLI · effort low · fixed exact reply, 5 runs",
          "bContext": "Codex CLI · effort high · fixed exact reply, 5 runs",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            3.44,
            4.1
          ],
          "bRange": [
            3.37,
            4.3
          ]
        },
        {
          "metric": "CLI vs API: time for a small coding task (Total time)",
          "aValue": 14.15,
          "bValue": 17.85,
          "unit": "seconds",
          "aDisplay": "14.2 s",
          "bDisplay": "17.9 s",
          "winner": "unclear",
          "basis": "Only 3 runs per side; the run ranges (fastest to slowest) do not overlap (GPT-6.1 Sol (Codex CLI) at low effort 13.0 s to 14.4 s; GPT-6.1 Sol (Codex CLI) at high effort 17.7 s to 22.4 s), but 3 runs cannot show a reliable difference. A range is not a confidence interval.",
          "studySlug": "cli-model-latency-tokens",
          "n": 3,
          "chartId": "cli-vs-api-small-coding-latency",
          "aN": 3,
          "bN": 3,
          "aContext": "Codex CLI · effort low · small coding task, 3 runs",
          "bContext": "Codex CLI · effort high · small coding task, 3 runs",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            13.02,
            14.41
          ],
          "bRange": [
            17.68,
            22.42
          ]
        },
        {
          "metric": "CLI vs API: time for a small coding task (First useful output)",
          "aValue": 13.6,
          "bValue": 17.27,
          "unit": "seconds",
          "aDisplay": "13.6 s",
          "bDisplay": "17.3 s",
          "winner": "unclear",
          "basis": "Only 3 runs per side; the run ranges (fastest to slowest) do not overlap (GPT-6.1 Sol (Codex CLI) at low effort 12.5 s to 13.8 s; GPT-6.1 Sol (Codex CLI) at high effort 17.1 s to 21.9 s), but 3 runs cannot show a reliable difference. A range is not a confidence interval.",
          "studySlug": "cli-model-latency-tokens",
          "n": 3,
          "chartId": "cli-vs-api-small-coding-latency",
          "aN": 3,
          "bN": 3,
          "aContext": "Codex CLI · effort low · small coding task, 3 runs",
          "bContext": "Codex CLI · effort high · small coding task, 3 runs",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            12.52,
            13.83
          ],
          "bRange": [
            17.13,
            21.86
          ]
        },
        {
          "metric": "Hidden prompt: input tokens for the same one-line request",
          "aValue": 19551,
          "bValue": 19555,
          "unit": "tokens",
          "aDisplay": "19,551",
          "bDisplay": "19,555",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "cli-model-latency-tokens",
          "n": 5,
          "chartId": "cli-vs-api-prompt-overhead",
          "aN": 5,
          "bN": 5,
          "aContext": "Codex CLI · effort low · short fixed tasks",
          "bContext": "Codex CLI · effort high · short fixed tasks"
        },
        {
          "metric": "Reasoning cost per strict pass by effort, with the total (calculation) (Reasoning cost per strict pass)",
          "aValue": 0.001223,
          "bValue": 0.004114,
          "unit": "usd",
          "aDisplay": "$0.0012",
          "bDisplay": "$0.0041",
          "winner": "unclear",
          "basis": "More or fewer usd is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "n": 16,
          "chartId": "thinking-bill-by-effort",
          "aN": 16,
          "bN": 16,
          "aContext": "Codex CLI · effort low",
          "bContext": "Codex CLI · effort high",
          "calculation": true
        },
        {
          "metric": "Reasoning cost per strict pass by effort, with the total (calculation) (Total cost per strict pass)",
          "aValue": 0.012837,
          "bValue": 0.015137,
          "unit": "usd",
          "aDisplay": "$0.013",
          "bDisplay": "$0.015",
          "winner": "unclear",
          "basis": "More or fewer usd is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "n": 16,
          "chartId": "thinking-bill-by-effort",
          "aN": 16,
          "bN": 16,
          "aContext": "Codex CLI · effort low",
          "bContext": "Codex CLI · effort high",
          "calculation": true
        }
      ]
    },
    {
      "slug": "gpt-6-1-sol-codex-cli-medium-vs-high",
      "entity": "gpt-6-1-sol-codex-cli",
      "aEffort": "medium",
      "bEffort": "high",
      "title": "GPT-6.1 Sol (Codex CLI): medium vs high effort",
      "seoTitle": "GPT-6.1 Sol (Codex CLI): medium vs high effort, measured",
      "description": "GPT-6.1 Sol (Codex CLI) at medium vs high effort: 14 measured metrics from 3 studies, with sample sizes, intervals and every failure counted.",
      "verdict": "GPT-6.1 Sol (Codex CLI) at medium effort and GPT-6.1 Sol (Codex CLI) at high effort share 14 measured metrics and 12 list-price calculations from 4 studies. Only the effort setting differs between the two sides of a row; the route and the task set are the same. No row separates them: every interval or run range overlaps, too few runs were recorded, no interval was recorded, or more is not better for that metric. The rows are 5 ties and 21 unclear; each row says why. Calculation rows are derived from list prices and recorded counts; they are not bills or runs. Some rows rest on small samples (n = 15 at the smallest).",
      "rows": [
        {
          "metric": "Pass rate on five validated tasks",
          "aValue": 1,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "100% (15/15)",
          "bDisplay": "100% (15/15)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (GPT-6.1 Sol (Codex CLI) at medium effort 80% to 100%; GPT-6.1 Sol (Codex CLI) at high effort 80% to 100%), so this sample cannot separate them.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-pass-rate",
          "aN": 15,
          "bN": 15,
          "aContext": "Codex CLI · effort medium · five short validated tasks",
          "bContext": "Codex CLI · effort high · five short validated tasks",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.7961,
            1
          ],
          "bRange": [
            0.7961,
            1
          ]
        },
        {
          "metric": "Total time per call",
          "aValue": 5.65,
          "bValue": 5.6,
          "unit": "seconds",
          "aDisplay": "5.65 s",
          "bDisplay": "5.60 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (GPT-6.1 Sol (Codex CLI) at medium effort 4.10 s to 25.5 s; GPT-6.1 Sol (Codex CLI) at high effort 4.05 s to 19.5 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-total-latency",
          "aN": 15,
          "bN": 15,
          "aContext": "Codex CLI · effort medium · five short validated tasks",
          "bContext": "Codex CLI · effort high · five short validated tasks",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            4.1,
            25.46
          ],
          "bRange": [
            4.05,
            19.52
          ]
        },
        {
          "metric": "Time to first useful output",
          "aValue": 5.05,
          "bValue": 5.32,
          "unit": "seconds",
          "aDisplay": "5.05 s",
          "bDisplay": "5.32 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (GPT-6.1 Sol (Codex CLI) at medium effort 3.36 s to 17.8 s; GPT-6.1 Sol (Codex CLI) at high effort 3.64 s to 16.4 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-first-useful-latency",
          "aN": 15,
          "bN": 15,
          "aContext": "Codex CLI · effort medium · five short validated tasks",
          "bContext": "Codex CLI · effort high · five short validated tasks",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            3.36,
            17.82
          ],
          "bRange": [
            3.64,
            16.37
          ]
        },
        {
          "metric": "Input tokens per call: what the CLI sends (Cache read)",
          "aValue": 5180,
          "bValue": 6716,
          "unit": "tokens",
          "aDisplay": "5,180",
          "bDisplay": "6,716",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-input-tokens",
          "aN": 15,
          "bN": 15,
          "aContext": "Codex CLI · effort medium · five short validated tasks",
          "bContext": "Codex CLI · effort high · five short validated tasks"
        },
        {
          "metric": "Input tokens per call: what the CLI sends (Other input)",
          "aValue": 6943,
          "bValue": 5406,
          "unit": "tokens",
          "aDisplay": "6,943",
          "bDisplay": "5,406",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-input-tokens",
          "aN": 15,
          "bN": 15,
          "aContext": "Codex CLI · effort medium · five short validated tasks",
          "bContext": "Codex CLI · effort high · five short validated tasks"
        },
        {
          "metric": "Output tokens per call (Output tokens)",
          "aValue": 42,
          "bValue": 42,
          "unit": "tokens",
          "aDisplay": "42",
          "bDisplay": "42",
          "winner": "tie",
          "basis": "Same value. More or fewer is not better by itself for this metric.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-output-tokens",
          "aN": 15,
          "bN": 15,
          "aContext": "Codex CLI · effort medium · five short validated tasks",
          "bContext": "Codex CLI · effort high · five short validated tasks"
        },
        {
          "metric": "List-price cost per call (calculation)",
          "aValue": 0.01018,
          "bValue": 0.01047,
          "unit": "usd",
          "aDisplay": "$0.010",
          "bDisplay": "$0.010",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (GPT-6.1 Sol (Codex CLI) at medium effort $0.0054 to $0.027; GPT-6.1 Sol (Codex CLI) at high effort $0.0066 to $0.028); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-list-price-per-call",
          "aN": 15,
          "bN": 15,
          "aContext": "Codex CLI · effort medium · five short validated tasks",
          "bContext": "Codex CLI · effort high · five short validated tasks",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            0.0054,
            0.02686
          ],
          "bRange": [
            0.0066,
            0.02812
          ],
          "calculation": true
        },
        {
          "metric": "List-price cost per passing answer (calculation)",
          "aValue": 0.01564,
          "bValue": 0.01322,
          "unit": "usd",
          "aDisplay": "$0.016",
          "bDisplay": "$0.013",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.016 vs $0.013) is not tested against run-to-run variation.",
          "studySlug": "model-head-to-head",
          "n": 15,
          "chartId": "h2h-cost-per-pass",
          "aN": 15,
          "bN": 15,
          "aContext": "Codex CLI · effort medium · five short validated tasks",
          "bContext": "Codex CLI · effort high · five short validated tasks",
          "calculation": true
        },
        {
          "metric": "Pass rate on eight hard tasks (Strict pass)",
          "aValue": 1,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "100% (16/16)",
          "bDisplay": "100% (16/16)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (GPT-6.1 Sol (Codex CLI) at medium effort 81% to 100%; GPT-6.1 Sol (Codex CLI) at high effort 81% to 100%), so this sample cannot separate them.",
          "studySlug": "hard-model-head-to-head",
          "n": 16,
          "chartId": "hard-h2h-pass-rate",
          "aN": 16,
          "bN": 16,
          "aContext": "Codex CLI · effort medium · eight hard validated tasks",
          "bContext": "Codex CLI · effort high · eight hard validated tasks",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.8064,
            1
          ],
          "bRange": [
            0.8064,
            1
          ]
        },
        {
          "metric": "Pass rate on eight hard tasks (Lenient (format misses counted))",
          "aValue": 1,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "100% (16/16)",
          "bDisplay": "100% (16/16)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (GPT-6.1 Sol (Codex CLI) at medium effort 81% to 100%; GPT-6.1 Sol (Codex CLI) at high effort 81% to 100%), so this sample cannot separate them.",
          "studySlug": "hard-model-head-to-head",
          "n": 16,
          "chartId": "hard-h2h-pass-rate",
          "aN": 16,
          "bN": 16,
          "aContext": "Codex CLI · effort medium · eight hard validated tasks",
          "bContext": "Codex CLI · effort high · eight hard validated tasks",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.8064,
            1
          ],
          "bRange": [
            0.8064,
            1
          ]
        },
        {
          "metric": "Total time per call on hard tasks (separate batches)",
          "aValue": 13.11,
          "bValue": 18.12,
          "unit": "seconds",
          "aDisplay": "13.1 s",
          "bDisplay": "18.1 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (GPT-6.1 Sol (Codex CLI) at medium effort 8.54 s to 61.6 s; GPT-6.1 Sol (Codex CLI) at high effort 11.7 s to 92.2 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "hard-model-head-to-head",
          "n": 16,
          "chartId": "hard-h2h-total-latency",
          "aN": 16,
          "bN": 16,
          "aContext": "Codex CLI · effort medium · eight hard validated tasks",
          "bContext": "Codex CLI · effort high · eight hard validated tasks",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            8.54,
            61.6
          ],
          "bRange": [
            11.67,
            92.21
          ]
        },
        {
          "metric": "Time to first useful output on hard tasks",
          "aValue": 10.23,
          "bValue": 12.69,
          "unit": "seconds",
          "aDisplay": "10.2 s",
          "bDisplay": "12.7 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (GPT-6.1 Sol (Codex CLI) at medium effort 6.09 s to 40.4 s; GPT-6.1 Sol (Codex CLI) at high effort 8.93 s to 75.9 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "hard-model-head-to-head",
          "n": 16,
          "chartId": "hard-h2h-first-useful-latency",
          "aN": 16,
          "bN": 16,
          "aContext": "Codex CLI · effort medium · eight hard validated tasks",
          "bContext": "Codex CLI · effort high · eight hard validated tasks",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            6.09,
            40.41
          ],
          "bRange": [
            8.93,
            75.91
          ]
        },
        {
          "metric": "Output tokens per call on hard tasks (Output tokens)",
          "aValue": 335,
          "bValue": 436,
          "unit": "tokens",
          "aDisplay": "335",
          "bDisplay": "436",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "hard-model-head-to-head",
          "n": 16,
          "chartId": "hard-h2h-output-tokens",
          "aN": 16,
          "bN": 16,
          "aContext": "Codex CLI · effort medium · eight hard validated tasks",
          "bContext": "Codex CLI · effort high · eight hard validated tasks"
        },
        {
          "metric": "List-price cost per strict pass on hard tasks (calculation)",
          "aValue": 0.02564,
          "bValue": 0.01514,
          "unit": "usd",
          "aDisplay": "$0.026",
          "bDisplay": "$0.015",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.026 vs $0.015) is not tested against run-to-run variation.",
          "studySlug": "hard-model-head-to-head",
          "n": 16,
          "chartId": "hard-h2h-cost-per-pass",
          "aN": 16,
          "bN": 16,
          "aContext": "Codex CLI · effort medium · eight hard validated tasks",
          "bContext": "Codex CLI · effort high · eight hard validated tasks",
          "calculation": true
        },
        {
          "metric": "Strict pass rate by effort on eight hard tasks",
          "aValue": 1,
          "bValue": 1,
          "unit": "rate",
          "aDisplay": "100% (16/16)",
          "bDisplay": "100% (16/16)",
          "winner": "tie",
          "basis": "The 95% intervals overlap (GPT-6.1 Sol (Codex CLI) at medium effort 81% to 100%; GPT-6.1 Sol (Codex CLI) at high effort 81% to 100%), so this sample cannot separate them.",
          "studySlug": "effort-ladder",
          "n": 16,
          "chartId": "effort-ladder-pass-rate",
          "aN": 16,
          "bN": 16,
          "aContext": "Codex CLI · effort medium · eight hard validated tasks, effort ladder",
          "bContext": "Codex CLI · effort high · eight hard validated tasks, effort ladder",
          "rangeKind": "ci95",
          "spanKind": "ci95",
          "aRange": [
            0.8064,
            1
          ],
          "bRange": [
            0.8064,
            1
          ]
        },
        {
          "metric": "Total time per call by effort on hard tasks",
          "aValue": 13.11,
          "bValue": 18.12,
          "unit": "seconds",
          "aDisplay": "13.1 s",
          "bDisplay": "18.1 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (GPT-6.1 Sol (Codex CLI) at medium effort 8.54 s to 61.6 s; GPT-6.1 Sol (Codex CLI) at high effort 11.7 s to 92.2 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "effort-ladder",
          "n": 16,
          "chartId": "effort-ladder-total-latency",
          "aN": 16,
          "bN": 16,
          "aContext": "Codex CLI · effort medium · eight hard validated tasks, effort ladder",
          "bContext": "Codex CLI · effort high · eight hard validated tasks, effort ladder",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            8.54,
            61.6
          ],
          "bRange": [
            11.67,
            92.21
          ]
        },
        {
          "metric": "Output tokens per call by effort on hard tasks (Output tokens)",
          "aValue": 335,
          "bValue": 436,
          "unit": "tokens",
          "aDisplay": "335",
          "bDisplay": "436",
          "winner": "unclear",
          "basis": "More or fewer tokens is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "effort-ladder",
          "n": 16,
          "chartId": "effort-ladder-output-tokens",
          "aN": 16,
          "bN": 16,
          "aContext": "Codex CLI · effort medium · eight hard validated tasks, effort ladder",
          "bContext": "Codex CLI · effort high · eight hard validated tasks, effort ladder"
        },
        {
          "metric": "List-price cost per strict pass by effort (calculation)",
          "aValue": 0.02564,
          "bValue": 0.01514,
          "unit": "usd",
          "aDisplay": "$0.026",
          "bDisplay": "$0.015",
          "winner": "unclear",
          "basis": "No interval or range was recorded for either side, so the gap ($0.026 vs $0.015) is not tested against run-to-run variation.",
          "studySlug": "effort-ladder",
          "n": 16,
          "chartId": "effort-ladder-cost-per-pass",
          "aN": 16,
          "bN": 16,
          "aContext": "Codex CLI · effort medium · eight hard validated tasks, effort ladder",
          "bContext": "Codex CLI · effort high · eight hard validated tasks, effort ladder",
          "calculation": true
        },
        {
          "metric": "Reasoning share of output tokens per call on hard tasks (calculation)",
          "aValue": 46.33,
          "bValue": 57.01,
          "unit": "percent",
          "aDisplay": "46.3%",
          "bDisplay": "57%",
          "winner": "unclear",
          "basis": "More or fewer percent is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "n": 16,
          "chartId": "thinking-bill-share",
          "aN": 16,
          "bN": 16,
          "aContext": "Codex CLI · effort medium",
          "bContext": "Codex CLI · effort high",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            11.42,
            86.85
          ],
          "bRange": [
            29.19,
            90.8
          ],
          "calculation": true
        },
        {
          "metric": "List-price cost per call: reasoning, remaining output and input (calculation) (Reasoning (output tokens))",
          "aValue": 0.002273,
          "bValue": 0.004114,
          "unit": "usd",
          "aDisplay": "$0.0023",
          "bDisplay": "$0.0041",
          "winner": "unclear",
          "basis": "More or fewer usd is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "n": 16,
          "chartId": "thinking-bill-cost-per-call",
          "aN": 16,
          "bN": 16,
          "aContext": "Codex CLI · effort medium",
          "bContext": "Codex CLI · effort high",
          "calculation": true
        },
        {
          "metric": "List-price cost per call: reasoning, remaining output and input (calculation) (Remaining output (visible-answer estimate))",
          "aValue": 0.003025,
          "bValue": 0.002905,
          "unit": "usd",
          "aDisplay": "$0.0030",
          "bDisplay": "$0.0029",
          "winner": "unclear",
          "basis": "More or fewer usd is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "n": 16,
          "chartId": "thinking-bill-cost-per-call",
          "aN": 16,
          "bN": 16,
          "aContext": "Codex CLI · effort medium",
          "bContext": "Codex CLI · effort high",
          "calculation": true
        },
        {
          "metric": "List-price cost per call: reasoning, remaining output and input (calculation) (Input (prompt, cache priced))",
          "aValue": 0.020339,
          "bValue": 0.008117,
          "unit": "usd",
          "aDisplay": "$0.020",
          "bDisplay": "$0.0081",
          "winner": "unclear",
          "basis": "More or fewer usd is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "n": 16,
          "chartId": "thinking-bill-cost-per-call",
          "aN": 16,
          "bN": 16,
          "aContext": "Codex CLI · effort medium",
          "bContext": "Codex CLI · effort high",
          "calculation": true
        },
        {
          "metric": "Reasoning cost per strict pass by effort, with the total (calculation) (Reasoning cost per strict pass)",
          "aValue": 0.002273,
          "bValue": 0.004114,
          "unit": "usd",
          "aDisplay": "$0.0023",
          "bDisplay": "$0.0041",
          "winner": "unclear",
          "basis": "More or fewer usd is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "n": 16,
          "chartId": "thinking-bill-by-effort",
          "aN": 16,
          "bN": 16,
          "aContext": "Codex CLI · effort medium",
          "bContext": "Codex CLI · effort high",
          "calculation": true
        },
        {
          "metric": "Reasoning cost per strict pass by effort, with the total (calculation) (Total cost per strict pass)",
          "aValue": 0.025637,
          "bValue": 0.015137,
          "unit": "usd",
          "aDisplay": "$0.026",
          "bDisplay": "$0.015",
          "winner": "unclear",
          "basis": "More or fewer usd is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "n": 16,
          "chartId": "thinking-bill-by-effort",
          "aN": 16,
          "bN": 16,
          "aContext": "Codex CLI · effort medium",
          "bContext": "Codex CLI · effort high",
          "calculation": true
        },
        {
          "metric": "Reasoning share on short tasks vs hard tasks (calculation) (Eight hard tasks)",
          "aValue": 46.33,
          "bValue": 57.01,
          "unit": "percent",
          "aDisplay": "46.3%",
          "bDisplay": "57%",
          "winner": "unclear",
          "basis": "More or fewer percent is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "n": 16,
          "chartId": "thinking-bill-short-vs-hard",
          "aN": 16,
          "bN": 16,
          "aContext": "Codex CLI · effort medium",
          "bContext": "Codex CLI · effort high",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            11.42,
            86.85
          ],
          "bRange": [
            29.19,
            90.8
          ],
          "calculation": true
        },
        {
          "metric": "Reasoning share on short tasks vs hard tasks (calculation) (Five short tasks)",
          "aValue": 41.05,
          "bValue": 58.06,
          "unit": "percent",
          "aDisplay": "41%",
          "bDisplay": "58.1%",
          "winner": "unclear",
          "basis": "More or fewer percent is not better or worse by itself; this row describes behaviour, not a winner.",
          "studySlug": "thinking-token-bill",
          "n": 15,
          "chartId": "thinking-bill-short-vs-hard",
          "aN": 15,
          "bN": 15,
          "aContext": "Codex CLI · effort medium",
          "bContext": "Codex CLI · effort high",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            0,
            71.43
          ],
          "bRange": [
            0,
            75.76
          ],
          "calculation": true
        }
      ]
    },
    {
      "slug": "gpt-6-1-sol-openai-api-low-vs-high",
      "entity": "gpt-6-1-sol-openai-api",
      "aEffort": "low",
      "bEffort": "high",
      "title": "GPT-6.1 Sol (OpenAI API): low vs high effort",
      "seoTitle": "GPT-6.1 Sol (OpenAI API): low vs high effort, measured",
      "description": "GPT-6.1 Sol (OpenAI API) at low vs high effort: 5 measured metrics from one study, with sample sizes, intervals and every failure counted.",
      "verdict": "GPT-6.1 Sol (OpenAI API) at low effort and GPT-6.1 Sol (OpenAI API) at high effort share 5 measured metrics from one study. Only the effort setting differs between the two sides of a row; the route and the task set are the same. No row separates them: every interval or run range overlaps, too few runs were recorded, no interval was recorded, or more is not better for that metric. The rows are 1 tie and 4 unclear; each row says why. Some rows rest on small samples (n = 3 at the smallest).",
      "rows": [
        {
          "metric": "CLI vs API: time for a one-line answer (Total time)",
          "aValue": 1.02,
          "bValue": 1.52,
          "unit": "seconds",
          "aDisplay": "1.02 s",
          "bDisplay": "1.52 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (GPT-6.1 Sol (OpenAI API) at low effort 0.96 s to 1.87 s; GPT-6.1 Sol (OpenAI API) at high effort 1.35 s to 2.23 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "cli-model-latency-tokens",
          "n": 5,
          "chartId": "cli-vs-api-exact-reply-latency",
          "aN": 5,
          "bN": 5,
          "aContext": "OpenAI API · effort low · fixed exact reply, 5 runs",
          "bContext": "OpenAI API · effort high · fixed exact reply, 5 runs",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            0.96,
            1.87
          ],
          "bRange": [
            1.35,
            2.23
          ]
        },
        {
          "metric": "CLI vs API: time for a one-line answer (First useful output)",
          "aValue": 0.87,
          "bValue": 1.34,
          "unit": "seconds",
          "aDisplay": "0.87 s",
          "bDisplay": "1.34 s",
          "winner": "unclear",
          "basis": "The run ranges (fastest to slowest) overlap (GPT-6.1 Sol (OpenAI API) at low effort 0.84 s to 1.74 s; GPT-6.1 Sol (OpenAI API) at high effort 1.26 s to 2.12 s); the medians alone do not show a reliable difference. A range is not a confidence interval.",
          "studySlug": "cli-model-latency-tokens",
          "n": 5,
          "chartId": "cli-vs-api-exact-reply-latency",
          "aN": 5,
          "bN": 5,
          "aContext": "OpenAI API · effort low · fixed exact reply, 5 runs",
          "bContext": "OpenAI API · effort high · fixed exact reply, 5 runs",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            0.84,
            1.74
          ],
          "bRange": [
            1.26,
            2.12
          ]
        },
        {
          "metric": "CLI vs API: time for a small coding task (Total time)",
          "aValue": 6,
          "bValue": 9.56,
          "unit": "seconds",
          "aDisplay": "6.00 s",
          "bDisplay": "9.56 s",
          "winner": "unclear",
          "basis": "Only 3 runs per side; the run ranges (fastest to slowest) do not overlap (GPT-6.1 Sol (OpenAI API) at low effort 5.44 s to 6.20 s; GPT-6.1 Sol (OpenAI API) at high effort 9.44 s to 10.9 s), but 3 runs cannot show a reliable difference. A range is not a confidence interval.",
          "studySlug": "cli-model-latency-tokens",
          "n": 3,
          "chartId": "cli-vs-api-small-coding-latency",
          "aN": 3,
          "bN": 3,
          "aContext": "OpenAI API · effort low · small coding task, 3 runs",
          "bContext": "OpenAI API · effort high · small coding task, 3 runs",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            5.44,
            6.2
          ],
          "bRange": [
            9.44,
            10.94
          ]
        },
        {
          "metric": "CLI vs API: time for a small coding task (First useful output)",
          "aValue": 1.05,
          "bValue": 5.31,
          "unit": "seconds",
          "aDisplay": "1.05 s",
          "bDisplay": "5.31 s",
          "winner": "unclear",
          "basis": "Only 3 runs per side; the run ranges (fastest to slowest) do not overlap (GPT-6.1 Sol (OpenAI API) at low effort 0.97 s to 1.40 s; GPT-6.1 Sol (OpenAI API) at high effort 4.99 s to 6.42 s), but 3 runs cannot show a reliable difference. A range is not a confidence interval.",
          "studySlug": "cli-model-latency-tokens",
          "n": 3,
          "chartId": "cli-vs-api-small-coding-latency",
          "aN": 3,
          "bN": 3,
          "aContext": "OpenAI API · effort low · small coding task, 3 runs",
          "bContext": "OpenAI API · effort high · small coding task, 3 runs",
          "rangeKind": "range",
          "spanKind": "minmax",
          "aRange": [
            0.97,
            1.4
          ],
          "bRange": [
            4.99,
            6.42
          ]
        },
        {
          "metric": "Hidden prompt: input tokens for the same one-line request",
          "aValue": 17,
          "bValue": 17,
          "unit": "tokens",
          "aDisplay": "17",
          "bDisplay": "17",
          "winner": "tie",
          "basis": "Same value. More or fewer is not better by itself for this metric.",
          "studySlug": "cli-model-latency-tokens",
          "n": 5,
          "chartId": "cli-vs-api-prompt-overhead",
          "aN": 5,
          "bN": 5,
          "aContext": "OpenAI API · effort low · short fixed tasks",
          "bContext": "OpenAI API · effort high · short fixed tasks"
        }
      ]
    }
  ],
  "leaderboard": {
    "note": "Per entity, the best-supported metrics across studies: a 95% interval first, then a run range, then the larger sample. Each keeps its study, n, interval or range and configuration. There is no composite score: the studies differ in task, route, effort and sample size, so one weighted number would hide what each value measured.",
    "perCategory": 3,
    "entries": [
      {
        "entity": "claude-sonnet-5-5",
        "name": "Claude Sonnet 5.5",
        "kind": "model",
        "vendor": "Anthropic",
        "measuredIn": 16,
        "studies": [
          "swe-bench-opus-vs-sonnet",
          "model-head-to-head",
          "hard-model-head-to-head",
          "coding-agents-head-to-head",
          "effort-ladder",
          "caching-consistency",
          "agent-memory",
          "routing-jev-vs-llm",
          "routing-overhead",
          "cli-model-latency-tokens",
          "single-call-vs-agent-loop",
          "haiku-thinking-on-off",
          "json-schema-vs-instructions",
          "prompt-cache-break-even",
          "routing-holdout",
          "thinking-token-bill",
          "llm-speed-anatomy",
          "haiku-retry-or-escalate",
          "harder-tasks-head-to-head"
        ],
        "factCount": 229,
        "best": [
          {
            "category": "quality",
            "studySlug": "routing-jev-vs-llm",
            "chartId": "routing-key-accuracy",
            "series": "Key accuracy",
            "point": "Claude Sonnet 5.5",
            "metric": "routing-key-accuracy",
            "label": "Per-question accuracy",
            "value": 0.9742,
            "unit": "rate",
            "display": "97% (189/194)",
            "n": 194,
            "ci": [
              0.9411,
              0.9889
            ],
            "spanKind": "ci95",
            "context": "typed routing decisions · via Claude Code"
          },
          {
            "category": "quality",
            "studySlug": "haiku-thinking-on-off",
            "chartId": "haiku-thinking-router-exact",
            "series": "Per-question accuracy",
            "point": "Claude Sonnet 5.5 (low) · Claude Code",
            "metric": "haiku-thinking-router-exact/Per-question accuracy",
            "label": "Haiku thinking study: typed routing decisions answered exactly right (Per-question accuracy)",
            "value": 0.9742,
            "unit": "rate",
            "display": "97% (189/194)",
            "n": 194,
            "ci": [
              0.9411,
              0.9889
            ],
            "spanKind": "ci95",
            "polarity": "higher",
            "context": "Claude Code · effort low · typed routing decisions, thinking on vs off"
          },
          {
            "category": "quality",
            "studySlug": "routing-holdout",
            "chartId": "routing-holdout-key-accuracy",
            "series": "Key accuracy",
            "point": "Claude Sonnet 5.5 (low) · Claude Code",
            "metric": "routing-holdout-key-accuracy",
            "label": "Per-question accuracy on unseen decisions",
            "value": 0.92,
            "unit": "rate",
            "display": "92% (115/125)",
            "n": 125,
            "ci": [
              0.859,
              0.956
            ],
            "spanKind": "ci95",
            "polarity": "higher",
            "context": "Claude Code · effort low"
          },
          {
            "category": "speed",
            "studySlug": "routing-jev-vs-llm",
            "chartId": "routing-decision-latency",
            "series": "Wall time (CLI)",
            "point": "Claude Sonnet 5.5",
            "metric": "routing-decision-latency/Wall time (CLI)",
            "label": "Time per routing decision (Wall time (CLI))",
            "value": 2598,
            "unit": "ms",
            "display": "2,598 ms",
            "n": 82,
            "range": [
              2598,
              4298
            ],
            "spanKind": "p50-p95",
            "context": "typed routing decisions · via Claude Code"
          },
          {
            "category": "speed",
            "studySlug": "routing-overhead",
            "chartId": "router-overhead-decision-latency",
            "series": "Decision time",
            "point": "Claude Sonnet 5.5 (effort low, via Claude Code)",
            "metric": "router-overhead-decision-latency",
            "label": "Time to make one routing decision",
            "value": 2597,
            "unit": "ms",
            "display": "2,597 ms",
            "n": 82,
            "range": [
              2597,
              4298
            ],
            "spanKind": "p50-p95",
            "context": "effort low · via Claude Code · routing overhead per decision"
          },
          {
            "category": "speed",
            "studySlug": "haiku-thinking-on-off",
            "chartId": "haiku-thinking-router-latency",
            "series": "Wall time (CLI)",
            "point": "Claude Sonnet 5.5 (low) · Claude Code",
            "metric": "haiku-thinking-router-latency/Wall time (CLI)",
            "label": "Haiku thinking study: time per routing decision (Wall time (CLI))",
            "value": 2.6,
            "unit": "seconds",
            "display": "2.60 s",
            "n": 82,
            "range": [
              2.6,
              4.3
            ],
            "spanKind": "p50-p95",
            "context": "Claude Code · effort low · typed routing decisions, thinking on vs off"
          },
          {
            "category": "cost",
            "studySlug": "model-head-to-head",
            "chartId": "h2h-list-price-per-call",
            "series": "Cost per call",
            "point": "Claude Sonnet 5.5 · Claude Code",
            "metric": "h2h-list-price-per-call",
            "label": "List-price cost per call (calculation)",
            "value": 0.0036,
            "unit": "usd",
            "display": "$0.0036",
            "n": 15,
            "range": [
              0.00342,
              0.01021
            ],
            "spanKind": "minmax",
            "calculation": true,
            "context": "Claude Code · five short validated tasks"
          },
          {
            "category": "cost",
            "studySlug": "routing-jev-vs-llm",
            "chartId": "routing-cost-per-1000",
            "series": "Cost",
            "point": "Claude Sonnet 5.5",
            "metric": "routing-cost-per-1000",
            "label": "Cost per 1,000 routing decisions",
            "value": 4.996,
            "unit": "usd",
            "display": "$5.00",
            "n": 82,
            "calculation": true,
            "context": "typed routing decisions · via Claude Code"
          },
          {
            "category": "cost",
            "studySlug": "haiku-thinking-on-off",
            "chartId": "haiku-thinking-router-cost",
            "series": "Cost per 1,000 decisions",
            "point": "Claude Sonnet 5.5 (low) · Claude Code",
            "metric": "haiku-thinking-router-cost",
            "label": "Haiku thinking study: list-price cost per 1,000 routing decisions (calculation)",
            "value": 7.324,
            "unit": "usd",
            "display": "$7.32",
            "n": 82,
            "calculation": true,
            "context": "Claude Code · effort low · typed routing decisions, thinking on vs off"
          }
        ]
      },
      {
        "entity": "claude-code-cli",
        "name": "Claude Code",
        "kind": "cli",
        "vendor": "Anthropic",
        "measuredIn": 14,
        "studies": [
          "model-head-to-head",
          "hard-model-head-to-head",
          "coding-agents-head-to-head",
          "effort-ladder",
          "caching-consistency",
          "agent-memory",
          "routing-overhead",
          "cli-model-latency-tokens",
          "single-call-vs-agent-loop",
          "haiku-thinking-on-off",
          "json-schema-vs-instructions",
          "routing-holdout",
          "thinking-token-bill",
          "llm-speed-anatomy",
          "haiku-retry-or-escalate",
          "harder-tasks-head-to-head"
        ],
        "factCount": 412,
        "best": [
          {
            "category": "quality",
            "studySlug": "haiku-thinking-on-off",
            "chartId": "haiku-thinking-router-exact",
            "series": "Per-question accuracy",
            "point": "Claude Sonnet 5.5 (low) · Claude Code",
            "metric": "haiku-thinking-router-exact/Per-question accuracy",
            "label": "Haiku thinking study: typed routing decisions answered exactly right (Per-question accuracy)",
            "value": 0.9742,
            "unit": "rate",
            "display": "97% (189/194)",
            "n": 194,
            "ci": [
              0.9411,
              0.9889
            ],
            "spanKind": "ci95",
            "polarity": "higher",
            "context": "Claude Sonnet 5.5 · effort low · typed routing decisions, thinking on vs off"
          },
          {
            "category": "quality",
            "studySlug": "routing-holdout",
            "chartId": "routing-holdout-key-accuracy",
            "series": "Key accuracy",
            "point": "Claude Sonnet 5.5 (low) · Claude Code",
            "metric": "routing-holdout-key-accuracy",
            "label": "Per-question accuracy on unseen decisions",
            "value": 0.92,
            "unit": "rate",
            "display": "92% (115/125)",
            "n": 125,
            "ci": [
              0.859,
              0.956
            ],
            "spanKind": "ci95",
            "polarity": "higher",
            "context": "Claude Sonnet 5.5 · effort low"
          },
          {
            "category": "quality",
            "studySlug": "hard-model-head-to-head",
            "chartId": "hard-h2h-pass-rate",
            "series": "Strict pass",
            "point": "Claude Sonnet 5.5 · Claude Code",
            "metric": "hard-h2h-pass-rate/Strict pass",
            "label": "Pass rate on eight hard tasks (Strict pass)",
            "value": 1,
            "unit": "rate",
            "display": "100% (24/24)",
            "n": 24,
            "ci": [
              0.862,
              1
            ],
            "spanKind": "ci95",
            "context": "Claude Sonnet 5.5 · eight hard validated tasks"
          },
          {
            "category": "speed",
            "studySlug": "haiku-thinking-on-off",
            "chartId": "haiku-thinking-router-latency",
            "series": "Wall time (CLI)",
            "point": "Claude Sonnet 5.5 (low) · Claude Code",
            "metric": "haiku-thinking-router-latency/Wall time (CLI)",
            "label": "Haiku thinking study: time per routing decision (Wall time (CLI))",
            "value": 2.6,
            "unit": "seconds",
            "display": "2.60 s",
            "n": 82,
            "range": [
              2.6,
              4.3
            ],
            "spanKind": "p50-p95",
            "context": "Claude Sonnet 5.5 · effort low · typed routing decisions, thinking on vs off"
          },
          {
            "category": "speed",
            "studySlug": "routing-holdout",
            "chartId": "routing-holdout-latency",
            "series": "Wall time",
            "point": "Claude Sonnet 5.5 (low) · Claude Code",
            "metric": "routing-holdout-latency/Wall time",
            "label": "Time per routing decision, by route (Wall time)",
            "value": 2.359,
            "unit": "seconds",
            "display": "2.36 s",
            "n": 56,
            "range": [
              2.359,
              3.657
            ],
            "spanKind": "p50-p95",
            "context": "Claude Sonnet 5.5 · effort low"
          },
          {
            "category": "speed",
            "studySlug": "hard-model-head-to-head",
            "chartId": "hard-h2h-total-latency",
            "series": "Total time per call on hard tasks (separate batches)",
            "point": "Claude Sonnet 5.5 · Claude Code",
            "metric": "hard-h2h-total-latency",
            "label": "Total time per call on hard tasks (separate batches)",
            "value": 7.75,
            "unit": "seconds",
            "display": "7.75 s",
            "n": 24,
            "range": [
              2.26,
              34.79
            ],
            "spanKind": "minmax",
            "context": "Claude Sonnet 5.5 · eight hard validated tasks"
          },
          {
            "category": "cost",
            "studySlug": "model-head-to-head",
            "chartId": "h2h-list-price-per-call",
            "series": "Cost per call",
            "point": "Claude Sonnet 5.5 · Claude Code",
            "metric": "h2h-list-price-per-call",
            "label": "List-price cost per call (calculation)",
            "value": 0.0036,
            "unit": "usd",
            "display": "$0.0036",
            "n": 15,
            "range": [
              0.00342,
              0.01021
            ],
            "spanKind": "minmax",
            "calculation": true,
            "context": "Claude Sonnet 5.5 · five short validated tasks"
          },
          {
            "category": "cost",
            "studySlug": "haiku-thinking-on-off",
            "chartId": "haiku-thinking-router-cost",
            "series": "Cost per 1,000 decisions",
            "point": "Claude Sonnet 5.5 (low) · Claude Code",
            "metric": "haiku-thinking-router-cost",
            "label": "Haiku thinking study: list-price cost per 1,000 routing decisions (calculation)",
            "value": 7.324,
            "unit": "usd",
            "display": "$7.32",
            "n": 82,
            "calculation": true,
            "context": "Claude Sonnet 5.5 · effort low · typed routing decisions, thinking on vs off"
          },
          {
            "category": "cost",
            "studySlug": "routing-holdout",
            "chartId": "routing-holdout-cost-per-1000",
            "series": "Cost",
            "point": "Claude Sonnet 5.5 (low) · Claude Code",
            "metric": "routing-holdout-cost-per-1000",
            "label": "Cost per 1,000 unseen routing decisions",
            "value": 7.244,
            "unit": "usd",
            "display": "$7.24",
            "n": 56,
            "calculation": true,
            "context": "Claude Sonnet 5.5 · effort low"
          }
        ]
      },
      {
        "entity": "claude-haiku-4-5",
        "name": "Claude Haiku 4.5",
        "kind": "model",
        "vendor": "Anthropic",
        "measuredIn": 14,
        "studies": [
          "swe-bench-verified",
          "model-head-to-head",
          "hard-model-head-to-head",
          "caching-consistency",
          "agent-memory",
          "routing-jev-vs-llm",
          "routing-overhead",
          "cost-thought-experiments",
          "single-call-vs-agent-loop",
          "haiku-thinking-on-off",
          "json-schema-vs-instructions",
          "routing-holdout",
          "thinking-token-bill",
          "llm-speed-anatomy",
          "haiku-retry-or-escalate",
          "harder-tasks-head-to-head"
        ],
        "factCount": 198,
        "best": [
          {
            "category": "quality",
            "studySlug": "routing-jev-vs-llm",
            "chartId": "routing-key-accuracy",
            "series": "Key accuracy",
            "point": "Claude Haiku 4.5",
            "metric": "routing-key-accuracy",
            "label": "Per-question accuracy",
            "value": 0.9433,
            "unit": "rate",
            "display": "94% (183/194)",
            "n": 194,
            "ci": [
              0.9013,
              0.968
            ],
            "spanKind": "ci95",
            "context": "typed routing decisions · via Claude Code"
          },
          {
            "category": "quality",
            "studySlug": "haiku-thinking-on-off",
            "chartId": "haiku-thinking-router-exact",
            "series": "Per-question accuracy",
            "point": "Claude Haiku 4.5 (thinking off) · Claude Code",
            "metric": "haiku-thinking-router-exact/Per-question accuracy",
            "label": "Haiku thinking study: typed routing decisions answered exactly right (Per-question accuracy)",
            "value": 0.9124,
            "unit": "rate",
            "display": "91% (177/194)",
            "n": 194,
            "ci": [
              0.8642,
              0.9446
            ],
            "spanKind": "ci95",
            "polarity": "higher",
            "context": "Claude Code · thinking off · typed routing decisions, thinking on vs off"
          },
          {
            "category": "quality",
            "studySlug": "routing-holdout",
            "chartId": "routing-holdout-key-accuracy",
            "series": "Key accuracy",
            "point": "Claude Haiku 4.5 · Claude Code",
            "metric": "routing-holdout-key-accuracy",
            "label": "Per-question accuracy on unseen decisions",
            "value": 0.816,
            "unit": "rate",
            "display": "82% (102/125)",
            "n": 125,
            "ci": [
              0.739,
              0.8741
            ],
            "spanKind": "ci95",
            "polarity": "higher",
            "context": "Claude Code"
          },
          {
            "category": "speed",
            "studySlug": "routing-jev-vs-llm",
            "chartId": "routing-decision-latency",
            "series": "Wall time (CLI)",
            "point": "Claude Haiku 4.5",
            "metric": "routing-decision-latency/Wall time (CLI)",
            "label": "Time per routing decision (Wall time (CLI))",
            "value": 12674,
            "unit": "ms",
            "display": "12,674 ms",
            "n": 82,
            "range": [
              12674,
              34413
            ],
            "spanKind": "p50-p95",
            "context": "typed routing decisions · via Claude Code"
          },
          {
            "category": "speed",
            "studySlug": "routing-overhead",
            "chartId": "router-overhead-decision-latency",
            "series": "Decision time",
            "point": "Claude Haiku 4.5 (thinking on, via Claude Code)",
            "metric": "router-overhead-decision-latency",
            "label": "Time to make one routing decision",
            "value": 12543,
            "unit": "ms",
            "display": "12,543 ms",
            "n": 82,
            "range": [
              12543,
              34481
            ],
            "spanKind": "p50-p95",
            "context": "thinking on · via Claude Code · routing overhead per decision"
          },
          {
            "category": "speed",
            "studySlug": "haiku-thinking-on-off",
            "chartId": "haiku-thinking-router-latency",
            "series": "Wall time (CLI)",
            "point": "Claude Haiku 4.5 (thinking off) · Claude Code",
            "metric": "haiku-thinking-router-latency/Wall time (CLI)",
            "label": "Haiku thinking study: time per routing decision (Wall time (CLI))",
            "value": 4.66,
            "unit": "seconds",
            "display": "4.66 s",
            "n": 82,
            "range": [
              4.66,
              8.18
            ],
            "spanKind": "p50-p95",
            "context": "Claude Code · thinking off · typed routing decisions, thinking on vs off"
          },
          {
            "category": "cost",
            "studySlug": "model-head-to-head",
            "chartId": "h2h-list-price-per-call",
            "series": "Cost per call",
            "point": "Claude Haiku 4.5 · Claude Code",
            "metric": "h2h-list-price-per-call",
            "label": "List-price cost per call (calculation)",
            "value": 0.00566,
            "unit": "usd",
            "display": "$0.0057",
            "n": 15,
            "range": [
              0.00513,
              0.01804
            ],
            "spanKind": "minmax",
            "calculation": true,
            "context": "Claude Code · five short validated tasks"
          },
          {
            "category": "cost",
            "studySlug": "routing-jev-vs-llm",
            "chartId": "routing-cost-per-1000",
            "series": "Cost",
            "point": "Claude Haiku 4.5",
            "metric": "routing-cost-per-1000",
            "label": "Cost per 1,000 routing decisions",
            "value": 8.924,
            "unit": "usd",
            "display": "$8.92",
            "n": 82,
            "calculation": true,
            "context": "typed routing decisions · via Claude Code"
          },
          {
            "category": "cost",
            "studySlug": "haiku-thinking-on-off",
            "chartId": "haiku-thinking-router-cost",
            "series": "Cost per 1,000 decisions",
            "point": "Claude Haiku 4.5 (thinking off) · Claude Code",
            "metric": "haiku-thinking-router-cost",
            "label": "Haiku thinking study: list-price cost per 1,000 routing decisions (calculation)",
            "value": 3.364,
            "unit": "usd",
            "display": "$3.36",
            "n": 82,
            "calculation": true,
            "context": "Claude Code · thinking off · typed routing decisions, thinking on vs off"
          }
        ]
      },
      {
        "entity": "codex-cli",
        "name": "Codex CLI",
        "kind": "cli",
        "vendor": "OpenAI",
        "measuredIn": 11,
        "studies": [
          "model-head-to-head",
          "hard-model-head-to-head",
          "coding-agents-head-to-head",
          "effort-ladder",
          "caching-consistency",
          "routing-overhead",
          "cli-model-latency-tokens",
          "single-call-vs-agent-loop",
          "json-schema-vs-instructions",
          "thinking-token-bill",
          "llm-speed-anatomy",
          "harder-tasks-head-to-head"
        ],
        "factCount": 167,
        "best": [
          {
            "category": "quality",
            "studySlug": "hard-model-head-to-head",
            "chartId": "hard-h2h-pass-rate",
            "series": "Strict pass",
            "point": "GPT-6.1 Sol (medium) · Codex CLI",
            "metric": "hard-h2h-pass-rate/Strict pass",
            "label": "Pass rate on eight hard tasks (Strict pass)",
            "value": 1,
            "unit": "rate",
            "display": "100% (16/16)",
            "n": 16,
            "ci": [
              0.8064,
              1
            ],
            "spanKind": "ci95",
            "context": "GPT-6.1 Sol · effort medium · eight hard validated tasks"
          },
          {
            "category": "quality",
            "studySlug": "effort-ladder",
            "chartId": "effort-ladder-pass-rate",
            "series": "Strict pass",
            "point": "GPT-6.1 Sol (medium) · Codex CLI",
            "metric": "effort-ladder-pass-rate",
            "label": "Strict pass rate by effort on eight hard tasks",
            "value": 1,
            "unit": "rate",
            "display": "100% (16/16)",
            "n": 16,
            "ci": [
              0.8064,
              1
            ],
            "spanKind": "ci95",
            "context": "GPT-6.1 Sol · effort medium · eight hard validated tasks, effort ladder"
          },
          {
            "category": "quality",
            "studySlug": "single-call-vs-agent-loop",
            "chartId": "agent-loop-pass-rate",
            "series": "Strict pass",
            "point": "GPT-6 Luna (single call) · Codex CLI",
            "metric": "agent-loop-pass-rate",
            "label": "Strict pass rate: single call vs agent loop on eight hard tasks",
            "value": 0.625,
            "unit": "rate",
            "display": "63% (10/16)",
            "n": 16,
            "ci": [
              0.3864,
              0.8152
            ],
            "spanKind": "ci95",
            "polarity": "higher",
            "context": "GPT-6 Luna · single call"
          },
          {
            "category": "speed",
            "studySlug": "hard-model-head-to-head",
            "chartId": "hard-h2h-total-latency",
            "series": "Total time per call on hard tasks (separate batches)",
            "point": "GPT-6.1 Sol (medium) · Codex CLI",
            "metric": "hard-h2h-total-latency",
            "label": "Total time per call on hard tasks (separate batches)",
            "value": 13.11,
            "unit": "seconds",
            "display": "13.1 s",
            "n": 16,
            "range": [
              8.54,
              61.6
            ],
            "spanKind": "minmax",
            "context": "GPT-6.1 Sol · effort medium · eight hard validated tasks"
          },
          {
            "category": "speed",
            "studySlug": "effort-ladder",
            "chartId": "effort-ladder-total-latency",
            "series": "Total time per call by effort on hard tasks",
            "point": "GPT-6.1 Sol (medium) · Codex CLI",
            "metric": "effort-ladder-total-latency",
            "label": "Total time per call by effort on hard tasks",
            "value": 13.11,
            "unit": "seconds",
            "display": "13.1 s",
            "n": 16,
            "range": [
              8.54,
              61.6
            ],
            "spanKind": "minmax",
            "context": "GPT-6.1 Sol · effort medium · eight hard validated tasks, effort ladder"
          },
          {
            "category": "speed",
            "studySlug": "single-call-vs-agent-loop",
            "chartId": "agent-loop-total-time",
            "series": "Total time per attempt",
            "point": "GPT-6 Luna (single call) · Codex CLI",
            "metric": "agent-loop-total-time",
            "label": "Total time per attempt: single call vs agent loop",
            "value": 5.16,
            "unit": "seconds",
            "display": "5.16 s",
            "n": 16,
            "range": [
              3.59,
              11.32
            ],
            "spanKind": "minmax",
            "context": "GPT-6 Luna · single call"
          },
          {
            "category": "cost",
            "studySlug": "model-head-to-head",
            "chartId": "h2h-list-price-per-call",
            "series": "Cost per call",
            "point": "GPT-6.1 Sol (medium) · Codex CLI",
            "metric": "h2h-list-price-per-call",
            "label": "List-price cost per call (calculation)",
            "value": 0.01018,
            "unit": "usd",
            "display": "$0.010",
            "n": 15,
            "range": [
              0.0054,
              0.02686
            ],
            "spanKind": "minmax",
            "calculation": true,
            "context": "GPT-6.1 Sol · effort medium · five short validated tasks"
          },
          {
            "category": "cost",
            "studySlug": "hard-model-head-to-head",
            "chartId": "hard-h2h-cost-per-pass",
            "series": "Cost per strict pass",
            "point": "GPT-6.1 Sol (medium) · Codex CLI",
            "metric": "hard-h2h-cost-per-pass",
            "label": "List-price cost per strict pass on hard tasks (calculation)",
            "value": 0.02564,
            "unit": "usd",
            "display": "$0.026",
            "n": 16,
            "calculation": true,
            "context": "GPT-6.1 Sol · effort medium · eight hard validated tasks"
          },
          {
            "category": "cost",
            "studySlug": "effort-ladder",
            "chartId": "effort-ladder-cost-per-pass",
            "series": "Cost per strict pass",
            "point": "GPT-6.1 Sol (medium) · Codex CLI",
            "metric": "effort-ladder-cost-per-pass",
            "label": "List-price cost per strict pass by effort (calculation)",
            "value": 0.02564,
            "unit": "usd",
            "display": "$0.026",
            "n": 16,
            "calculation": true,
            "context": "GPT-6.1 Sol · effort medium · eight hard validated tasks, effort ladder"
          }
        ]
      },
      {
        "entity": "gpt-6-1-sol-codex-cli",
        "name": "GPT-6.1 Sol (Codex CLI)",
        "kind": "model",
        "vendor": "OpenAI",
        "measuredIn": 9,
        "studies": [
          "model-head-to-head",
          "hard-model-head-to-head",
          "coding-agents-head-to-head",
          "effort-ladder",
          "caching-consistency",
          "cli-model-latency-tokens",
          "json-schema-vs-instructions",
          "thinking-token-bill",
          "llm-speed-anatomy",
          "harder-tasks-head-to-head"
        ],
        "factCount": 128,
        "best": [
          {
            "category": "quality",
            "studySlug": "hard-model-head-to-head",
            "chartId": "hard-h2h-pass-rate",
            "series": "Strict pass",
            "point": "GPT-6.1 Sol (medium) · Codex CLI",
            "metric": "hard-h2h-pass-rate/Strict pass",
            "label": "Pass rate on eight hard tasks (Strict pass)",
            "value": 1,
            "unit": "rate",
            "display": "100% (16/16)",
            "n": 16,
            "ci": [
              0.8064,
              1
            ],
            "spanKind": "ci95",
            "context": "Codex CLI · effort medium · eight hard validated tasks"
          },
          {
            "category": "quality",
            "studySlug": "effort-ladder",
            "chartId": "effort-ladder-pass-rate",
            "series": "Strict pass",
            "point": "GPT-6.1 Sol (medium) · Codex CLI",
            "metric": "effort-ladder-pass-rate",
            "label": "Strict pass rate by effort on eight hard tasks",
            "value": 1,
            "unit": "rate",
            "display": "100% (16/16)",
            "n": 16,
            "ci": [
              0.8064,
              1
            ],
            "spanKind": "ci95",
            "context": "Codex CLI · effort medium · eight hard validated tasks, effort ladder"
          },
          {
            "category": "quality",
            "studySlug": "harder-tasks-head-to-head",
            "chartId": "harder-h2h-pass-rate",
            "series": "Strict pass",
            "point": "GPT-6.1 Sol (medium) · Codex CLI",
            "metric": "harder-h2h-pass-rate/Strict pass",
            "label": "Pass rate on 4 harder tasks (Strict pass)",
            "value": 0.6875,
            "unit": "rate",
            "display": "69% (11/16)",
            "n": 16,
            "ci": [
              0.444,
              0.8584
            ],
            "spanKind": "ci95",
            "polarity": "higher",
            "context": "Codex CLI · effort medium"
          },
          {
            "category": "speed",
            "studySlug": "hard-model-head-to-head",
            "chartId": "hard-h2h-total-latency",
            "series": "Total time per call on hard tasks (separate batches)",
            "point": "GPT-6.1 Sol (medium) · Codex CLI",
            "metric": "hard-h2h-total-latency",
            "label": "Total time per call on hard tasks (separate batches)",
            "value": 13.11,
            "unit": "seconds",
            "display": "13.1 s",
            "n": 16,
            "range": [
              8.54,
              61.6
            ],
            "spanKind": "minmax",
            "context": "Codex CLI · effort medium · eight hard validated tasks"
          },
          {
            "category": "speed",
            "studySlug": "effort-ladder",
            "chartId": "effort-ladder-total-latency",
            "series": "Total time per call by effort on hard tasks",
            "point": "GPT-6.1 Sol (medium) · Codex CLI",
            "metric": "effort-ladder-total-latency",
            "label": "Total time per call by effort on hard tasks",
            "value": 13.11,
            "unit": "seconds",
            "display": "13.1 s",
            "n": 16,
            "range": [
              8.54,
              61.6
            ],
            "spanKind": "minmax",
            "context": "Codex CLI · effort medium · eight hard validated tasks, effort ladder"
          },
          {
            "category": "speed",
            "studySlug": "model-head-to-head",
            "chartId": "h2h-total-latency",
            "series": "Total time per call",
            "point": "GPT-6.1 Sol (medium) · Codex CLI",
            "metric": "h2h-total-latency",
            "label": "Total time per call",
            "value": 5.65,
            "unit": "seconds",
            "display": "5.65 s",
            "n": 15,
            "range": [
              4.1,
              25.46
            ],
            "spanKind": "minmax",
            "context": "Codex CLI · effort medium · five short validated tasks"
          },
          {
            "category": "cost",
            "studySlug": "model-head-to-head",
            "chartId": "h2h-list-price-per-call",
            "series": "Cost per call",
            "point": "GPT-6.1 Sol (medium) · Codex CLI",
            "metric": "h2h-list-price-per-call",
            "label": "List-price cost per call (calculation)",
            "value": 0.01018,
            "unit": "usd",
            "display": "$0.010",
            "n": 15,
            "range": [
              0.0054,
              0.02686
            ],
            "spanKind": "minmax",
            "calculation": true,
            "context": "Codex CLI · effort medium · five short validated tasks"
          },
          {
            "category": "cost",
            "studySlug": "hard-model-head-to-head",
            "chartId": "hard-h2h-cost-per-pass",
            "series": "Cost per strict pass",
            "point": "GPT-6.1 Sol (medium) · Codex CLI",
            "metric": "hard-h2h-cost-per-pass",
            "label": "List-price cost per strict pass on hard tasks (calculation)",
            "value": 0.02564,
            "unit": "usd",
            "display": "$0.026",
            "n": 16,
            "calculation": true,
            "context": "Codex CLI · effort medium · eight hard validated tasks"
          },
          {
            "category": "cost",
            "studySlug": "effort-ladder",
            "chartId": "effort-ladder-cost-per-pass",
            "series": "Cost per strict pass",
            "point": "GPT-6.1 Sol (medium) · Codex CLI",
            "metric": "effort-ladder-cost-per-pass",
            "label": "List-price cost per strict pass by effort (calculation)",
            "value": 0.02564,
            "unit": "usd",
            "display": "$0.026",
            "n": 16,
            "calculation": true,
            "context": "Codex CLI · effort medium · eight hard validated tasks, effort ladder"
          }
        ]
      },
      {
        "entity": "claude-opus-5-5",
        "name": "Claude Opus 5.5",
        "kind": "model",
        "vendor": "Anthropic",
        "measuredIn": 8,
        "studies": [
          "swe-bench-opus-vs-sonnet",
          "model-head-to-head",
          "hard-model-head-to-head",
          "coding-agents-head-to-head",
          "effort-ladder",
          "caching-consistency",
          "prompt-cache-break-even",
          "thinking-token-bill",
          "llm-speed-anatomy",
          "harder-tasks-head-to-head"
        ],
        "factCount": 127,
        "best": [
          {
            "category": "quality",
            "studySlug": "hard-model-head-to-head",
            "chartId": "hard-h2h-pass-rate",
            "series": "Strict pass",
            "point": "Claude Opus 5.5 · Claude Code",
            "metric": "hard-h2h-pass-rate/Strict pass",
            "label": "Pass rate on eight hard tasks (Strict pass)",
            "value": 1,
            "unit": "rate",
            "display": "100% (24/24)",
            "n": 24,
            "ci": [
              0.862,
              1
            ],
            "spanKind": "ci95",
            "context": "Claude Code · eight hard validated tasks"
          },
          {
            "category": "quality",
            "studySlug": "effort-ladder",
            "chartId": "effort-ladder-pass-rate",
            "series": "Strict pass",
            "point": "Claude Opus 5.5 · Claude Code",
            "metric": "effort-ladder-pass-rate",
            "label": "Strict pass rate by effort on eight hard tasks",
            "value": 1,
            "unit": "rate",
            "display": "100% (16/16)",
            "n": 16,
            "ci": [
              0.8064,
              1
            ],
            "spanKind": "ci95",
            "context": "Claude Code · eight hard validated tasks, effort ladder"
          },
          {
            "category": "quality",
            "studySlug": "model-head-to-head",
            "chartId": "h2h-pass-rate",
            "series": "Pass rate",
            "point": "Claude Opus 5.5 · Claude Code",
            "metric": "h2h-pass-rate",
            "label": "Pass rate on five validated tasks",
            "value": 1,
            "unit": "rate",
            "display": "100% (15/15)",
            "n": 15,
            "ci": [
              0.7961,
              1
            ],
            "spanKind": "ci95",
            "context": "Claude Code · five short validated tasks"
          },
          {
            "category": "speed",
            "studySlug": "hard-model-head-to-head",
            "chartId": "hard-h2h-total-latency",
            "series": "Total time per call on hard tasks (separate batches)",
            "point": "Claude Opus 5.5 · Claude Code",
            "metric": "hard-h2h-total-latency",
            "label": "Total time per call on hard tasks (separate batches)",
            "value": 9.18,
            "unit": "seconds",
            "display": "9.18 s",
            "n": 24,
            "range": [
              4.24,
              27.21
            ],
            "spanKind": "minmax",
            "context": "Claude Code · eight hard validated tasks"
          },
          {
            "category": "speed",
            "studySlug": "effort-ladder",
            "chartId": "effort-ladder-total-latency",
            "series": "Total time per call by effort on hard tasks",
            "point": "Claude Opus 5.5 · Claude Code",
            "metric": "effort-ladder-total-latency",
            "label": "Total time per call by effort on hard tasks",
            "value": 9.18,
            "unit": "seconds",
            "display": "9.18 s",
            "n": 16,
            "range": [
              4.24,
              27.21
            ],
            "spanKind": "minmax",
            "context": "Claude Code · eight hard validated tasks, effort ladder"
          },
          {
            "category": "speed",
            "studySlug": "model-head-to-head",
            "chartId": "h2h-total-latency",
            "series": "Total time per call",
            "point": "Claude Opus 5.5 · Claude Code",
            "metric": "h2h-total-latency",
            "label": "Total time per call",
            "value": 2.75,
            "unit": "seconds",
            "display": "2.75 s",
            "n": 15,
            "range": [
              2.47,
              8.91
            ],
            "spanKind": "minmax",
            "context": "Claude Code · five short validated tasks"
          },
          {
            "category": "cost",
            "studySlug": "model-head-to-head",
            "chartId": "h2h-list-price-per-call",
            "series": "Cost per call",
            "point": "Claude Opus 5.5 · Claude Code",
            "metric": "h2h-list-price-per-call",
            "label": "List-price cost per call (calculation)",
            "value": 0.00688,
            "unit": "usd",
            "display": "$0.0069",
            "n": 15,
            "range": [
              0.00592,
              0.02226
            ],
            "spanKind": "minmax",
            "calculation": true,
            "context": "Claude Code · five short validated tasks"
          },
          {
            "category": "cost",
            "studySlug": "hard-model-head-to-head",
            "chartId": "hard-h2h-cost-per-pass",
            "series": "Cost per strict pass",
            "point": "Claude Opus 5.5 · Claude Code",
            "metric": "hard-h2h-cost-per-pass",
            "label": "List-price cost per strict pass on hard tasks (calculation)",
            "value": 0.02824,
            "unit": "usd",
            "display": "$0.028",
            "n": 24,
            "calculation": true,
            "context": "Claude Code · eight hard validated tasks"
          },
          {
            "category": "cost",
            "studySlug": "thinking-token-bill",
            "chartId": "thinking-bill-cost-per-call",
            "series": "Reasoning (output tokens)",
            "point": "Claude Opus 5.5 · Claude Code",
            "metric": "thinking-bill-cost-per-call/Reasoning (output tokens)",
            "label": "List-price cost per call: reasoning, remaining output and input (calculation) (Reasoning (output tokens))",
            "value": 0.012528,
            "unit": "usd",
            "display": "$0.013",
            "n": 24,
            "calculation": true,
            "polarity": "none",
            "context": "Claude Code"
          }
        ]
      },
      {
        "entity": "jev-1-13",
        "name": "Jev 1.13",
        "kind": "router",
        "vendor": "TypeSafe",
        "measuredIn": 4,
        "studies": [
          "system-one-arena",
          "routing-jev-vs-llm",
          "routing-overhead",
          "routing-holdout"
        ],
        "factCount": 57,
        "best": [
          {
            "category": "quality",
            "studySlug": "system-one-arena",
            "chartId": "arena-accuracy",
            "series": "Accuracy",
            "point": "Jev 1.13",
            "metric": "arena-accuracy",
            "label": "Who decides right? Accuracy on 1,000+ checkable decisions",
            "value": 0.7681,
            "unit": "rate",
            "display": "77% (805/1048)",
            "n": 1048,
            "ci": [
              0.7416,
              0.7927
            ],
            "spanKind": "ci95",
            "context": ""
          },
          {
            "category": "quality",
            "studySlug": "routing-overhead",
            "chartId": "router-overhead-completed",
            "series": "Completed",
            "point": "Jev 1.13 (TypeSafe)",
            "metric": "router-overhead-completed",
            "label": "Routing calls that returned a decision",
            "value": 1,
            "unit": "rate",
            "display": "100% (246/246)",
            "n": 246,
            "ci": [
              0.9846,
              1
            ],
            "spanKind": "ci95",
            "context": "routing overhead per decision · TypeSafe API"
          },
          {
            "category": "quality",
            "studySlug": "routing-jev-vs-llm",
            "chartId": "routing-key-accuracy",
            "series": "Key accuracy",
            "point": "Jev 1.13 (TypeSafe)",
            "metric": "routing-key-accuracy",
            "label": "Per-question accuracy",
            "value": 0.9485,
            "unit": "rate",
            "display": "95% (184/194)",
            "n": 194,
            "ci": [
              0.9077,
              0.9718
            ],
            "spanKind": "ci95",
            "context": "typed routing decisions · TypeSafe API"
          },
          {
            "category": "speed",
            "studySlug": "routing-jev-vs-llm",
            "chartId": "routing-decision-latency",
            "series": "Wall time (direct API call)",
            "point": "Jev 1.13 (TypeSafe)",
            "metric": "routing-decision-latency/Wall time (direct API call)",
            "label": "Time per routing decision (Wall time (direct API call))",
            "value": 136.5,
            "unit": "ms",
            "display": "137 ms",
            "n": 246,
            "range": [
              136.5,
              195.7
            ],
            "spanKind": "p50-p95",
            "context": "typed routing decisions · TypeSafe API"
          },
          {
            "category": "speed",
            "studySlug": "routing-overhead",
            "chartId": "router-overhead-decision-latency",
            "series": "Decision time",
            "point": "Jev 1.13 (TypeSafe)",
            "metric": "router-overhead-decision-latency",
            "label": "Time to make one routing decision",
            "value": 136.5,
            "unit": "ms",
            "display": "137 ms",
            "n": 246,
            "range": [
              136.5,
              195.7
            ],
            "spanKind": "p50-p95",
            "context": "routing overhead per decision · TypeSafe API"
          },
          {
            "category": "speed",
            "studySlug": "routing-holdout",
            "chartId": "routing-holdout-latency",
            "series": "Wall time",
            "point": "Jev 1.13 (TypeSafe)",
            "metric": "routing-holdout-latency/Wall time",
            "label": "Time per routing decision, by route (Wall time)",
            "value": 0.139,
            "unit": "seconds",
            "display": "0.14 s",
            "n": 168,
            "range": [
              0.139,
              0.192
            ],
            "spanKind": "p50-p95",
            "context": ""
          },
          {
            "category": "cost",
            "studySlug": "routing-jev-vs-llm",
            "chartId": "routing-cost-per-1000",
            "series": "Cost",
            "point": "Jev 1.13 (TypeSafe)",
            "metric": "routing-cost-per-1000",
            "label": "Cost per 1,000 routing decisions",
            "value": 0.0337,
            "unit": "usd",
            "display": "$0.034",
            "n": 246,
            "calculation": true,
            "context": "typed routing decisions · TypeSafe API"
          },
          {
            "category": "cost",
            "studySlug": "routing-holdout",
            "chartId": "routing-holdout-cost-per-1000",
            "series": "Cost",
            "point": "Jev 1.13 (TypeSafe)",
            "metric": "routing-holdout-cost-per-1000",
            "label": "Cost per 1,000 unseen routing decisions",
            "value": 0.03065,
            "unit": "usd",
            "display": "$0.031",
            "n": 168,
            "calculation": true,
            "context": ""
          },
          {
            "category": "cost",
            "studySlug": "routing-overhead",
            "chartId": "router-overhead-cost-reported",
            "series": "Cost per 1,000 decisions",
            "point": "Jev 1.13 (TypeSafe)",
            "metric": "router-overhead-cost-reported",
            "label": "Cost per 1,000 routing decisions: no model call vs provider-reported",
            "value": 0.0337,
            "unit": "usd",
            "display": "$0.034",
            "n": 82,
            "context": "routing overhead per decision · TypeSafe API"
          }
        ]
      },
      {
        "entity": "gpt-6-luna-codex-cli",
        "name": "GPT-6 Luna (Codex CLI)",
        "kind": "model",
        "vendor": "OpenAI",
        "measuredIn": 3,
        "studies": [
          "cli-model-latency-tokens",
          "single-call-vs-agent-loop",
          "llm-speed-anatomy"
        ],
        "factCount": 34,
        "best": [
          {
            "category": "quality",
            "studySlug": "single-call-vs-agent-loop",
            "chartId": "agent-loop-pass-rate",
            "series": "Strict pass",
            "point": "GPT-6 Luna (single call) · Codex CLI",
            "metric": "agent-loop-pass-rate",
            "label": "Strict pass rate: single call vs agent loop on eight hard tasks",
            "value": 0.625,
            "unit": "rate",
            "display": "63% (10/16)",
            "n": 16,
            "ci": [
              0.3864,
              0.8152
            ],
            "spanKind": "ci95",
            "polarity": "higher",
            "context": "Codex CLI · single call"
          },
          {
            "category": "speed",
            "studySlug": "single-call-vs-agent-loop",
            "chartId": "agent-loop-total-time",
            "series": "Total time per attempt",
            "point": "GPT-6 Luna (single call) · Codex CLI",
            "metric": "agent-loop-total-time",
            "label": "Total time per attempt: single call vs agent loop",
            "value": 5.16,
            "unit": "seconds",
            "display": "5.16 s",
            "n": 16,
            "range": [
              3.59,
              11.32
            ],
            "spanKind": "minmax",
            "context": "Codex CLI · single call"
          },
          {
            "category": "speed",
            "studySlug": "cli-model-latency-tokens",
            "chartId": "cli-vs-api-exact-reply-latency",
            "series": "Total time",
            "point": "Codex CLI · GPT-6 Luna · none",
            "metric": "cli-vs-api-exact-reply-latency/Total time",
            "label": "CLI vs API: time for a one-line answer (Total time)",
            "value": 3.19,
            "unit": "seconds",
            "display": "3.19 s",
            "n": 5,
            "range": [
              2.88,
              3.83
            ],
            "spanKind": "minmax",
            "context": "Codex CLI · effort none · fixed exact reply, 5 runs"
          },
          {
            "category": "speed",
            "studySlug": "llm-speed-anatomy",
            "chartId": "speed-anatomy-first-text",
            "series": "Time to first text",
            "point": "GPT-6 Luna (low) · Codex CLI",
            "metric": "speed-anatomy-first-text",
            "label": "Time to first text: a 250-line answer, six models",
            "value": 3.3,
            "unit": "seconds",
            "display": "3.30 s",
            "n": 4,
            "range": [
              3.19,
              3.47
            ],
            "spanKind": "minmax",
            "context": "Codex CLI · effort low"
          },
          {
            "category": "cost",
            "studySlug": "single-call-vs-agent-loop",
            "chartId": "agent-loop-cost-per-pass",
            "series": "Cost per strict pass",
            "point": "GPT-6 Luna (single call) · Codex CLI",
            "metric": "agent-loop-cost-per-pass",
            "label": "List-price cost per strict pass: single call vs agent loop (calculation)",
            "value": 0.00116,
            "unit": "usd",
            "display": "$0.0012",
            "n": 16,
            "calculation": true,
            "context": "Codex CLI · single call"
          }
        ]
      },
      {
        "entity": "claude-fable-5-1",
        "name": "Claude Fable 5.1",
        "kind": "model",
        "vendor": "Anthropic",
        "measuredIn": 3,
        "studies": [
          "model-head-to-head",
          "hard-model-head-to-head",
          "thinking-token-bill",
          "llm-speed-anatomy"
        ],
        "factCount": 23,
        "best": [
          {
            "category": "quality",
            "studySlug": "hard-model-head-to-head",
            "chartId": "hard-h2h-pass-rate",
            "series": "Strict pass",
            "point": "Claude Fable 5.1 · Claude Code",
            "metric": "hard-h2h-pass-rate/Strict pass",
            "label": "Pass rate on eight hard tasks (Strict pass)",
            "value": 1,
            "unit": "rate",
            "display": "100% (24/24)",
            "n": 24,
            "ci": [
              0.862,
              1
            ],
            "spanKind": "ci95",
            "context": "Claude Code · eight hard validated tasks"
          },
          {
            "category": "quality",
            "studySlug": "model-head-to-head",
            "chartId": "h2h-pass-rate",
            "series": "Pass rate",
            "point": "Claude Fable 5.1 · Claude Code",
            "metric": "h2h-pass-rate",
            "label": "Pass rate on five validated tasks",
            "value": 1,
            "unit": "rate",
            "display": "100% (15/15)",
            "n": 15,
            "ci": [
              0.7961,
              1
            ],
            "spanKind": "ci95",
            "context": "Claude Code · five short validated tasks"
          },
          {
            "category": "quality",
            "studySlug": "thinking-token-bill",
            "chartId": "thinking-bill-share",
            "series": "Median call",
            "point": "Claude Fable 5.1 · Claude Code",
            "metric": "thinking-bill-share",
            "label": "Reasoning share of output tokens per call on hard tasks (calculation)",
            "value": 64.24,
            "unit": "percent",
            "display": "64.2%",
            "n": 24,
            "range": [
              23.44,
              97.19
            ],
            "spanKind": "minmax",
            "calculation": true,
            "polarity": "none",
            "context": "Claude Code"
          },
          {
            "category": "speed",
            "studySlug": "hard-model-head-to-head",
            "chartId": "hard-h2h-total-latency",
            "series": "Total time per call on hard tasks (separate batches)",
            "point": "Claude Fable 5.1 · Claude Code",
            "metric": "hard-h2h-total-latency",
            "label": "Total time per call on hard tasks (separate batches)",
            "value": 16.13,
            "unit": "seconds",
            "display": "16.1 s",
            "n": 24,
            "range": [
              4.46,
              90
            ],
            "spanKind": "minmax",
            "context": "Claude Code · eight hard validated tasks"
          },
          {
            "category": "speed",
            "studySlug": "model-head-to-head",
            "chartId": "h2h-total-latency",
            "series": "Total time per call",
            "point": "Claude Fable 5.1 · Claude Code",
            "metric": "h2h-total-latency",
            "label": "Total time per call",
            "value": 1.94,
            "unit": "seconds",
            "display": "1.94 s",
            "n": 15,
            "range": [
              1.41,
              9.83
            ],
            "spanKind": "minmax",
            "context": "Claude Code · five short validated tasks"
          },
          {
            "category": "speed",
            "studySlug": "llm-speed-anatomy",
            "chartId": "speed-anatomy-first-text",
            "series": "Time to first text",
            "point": "Claude Fable 5.1 · Claude Code",
            "metric": "speed-anatomy-first-text",
            "label": "Time to first text: a 250-line answer, six models",
            "value": 4.43,
            "unit": "seconds",
            "display": "4.43 s",
            "n": 4,
            "range": [
              2.27,
              4.64
            ],
            "spanKind": "minmax",
            "context": "Claude Code"
          },
          {
            "category": "cost",
            "studySlug": "model-head-to-head",
            "chartId": "h2h-list-price-per-call",
            "series": "Cost per call",
            "point": "Claude Fable 5.1 · Claude Code",
            "metric": "h2h-list-price-per-call",
            "label": "List-price cost per call (calculation)",
            "value": 0.00987,
            "unit": "usd",
            "display": "$0.0099",
            "n": 15,
            "range": [
              0.0049,
              0.05843
            ],
            "spanKind": "minmax",
            "calculation": true,
            "context": "Claude Code · five short validated tasks"
          },
          {
            "category": "cost",
            "studySlug": "hard-model-head-to-head",
            "chartId": "hard-h2h-cost-per-pass",
            "series": "Cost per strict pass",
            "point": "Claude Fable 5.1 · Claude Code",
            "metric": "hard-h2h-cost-per-pass",
            "label": "List-price cost per strict pass on hard tasks (calculation)",
            "value": 0.09331,
            "unit": "usd",
            "display": "$0.093",
            "n": 24,
            "calculation": true,
            "context": "Claude Code · eight hard validated tasks"
          },
          {
            "category": "cost",
            "studySlug": "thinking-token-bill",
            "chartId": "thinking-bill-cost-per-call",
            "series": "Reasoning (output tokens)",
            "point": "Claude Fable 5.1 · Claude Code",
            "metric": "thinking-bill-cost-per-call/Reasoning (output tokens)",
            "label": "List-price cost per call: reasoning, remaining output and input (calculation) (Reasoning (output tokens))",
            "value": 0.053696,
            "unit": "usd",
            "display": "$0.054",
            "n": 24,
            "calculation": true,
            "polarity": "none",
            "context": "Claude Code"
          }
        ]
      },
      {
        "entity": "agent-harness",
        "name": "Agent",
        "kind": "harness",
        "vendor": "Agent",
        "measuredIn": 3,
        "studies": [
          "swe-bench-verified",
          "blind-review-head-to-head",
          "cost-thought-experiments",
          "coding-calibration"
        ],
        "factCount": 23,
        "best": [
          {
            "category": "quality",
            "studySlug": "blind-review-head-to-head",
            "statId": "verdicts-ai",
            "metric": "stat:verdicts-ai",
            "label": "Single critic verdicts that preferred the AI change",
            "value": 0.6894,
            "unit": "rate",
            "display": "69% (91/132)",
            "n": 132,
            "ci": [
              0.606,
              0.762
            ],
            "spanKind": "ci95",
            "context": "blind panel: Agent change vs merged human change · blind review panel"
          },
          {
            "category": "quality",
            "studySlug": "swe-bench-verified",
            "chartId": "swebench-same-instance-leaderboard",
            "series": "Resolved rate",
            "point": "Agent (Sonnet 5.5, full pipeline)",
            "metric": "swebench-same-instance-leaderboard",
            "label": "Resolved rate on the same 33 SWE-bench Verified instances",
            "value": 0.7576,
            "unit": "rate",
            "display": "76% (25/33)",
            "n": 33,
            "ci": [
              0.5898,
              0.8717
            ],
            "spanKind": "ci95",
            "context": "full pipeline on Claude Sonnet 5.5"
          },
          {
            "category": "speed",
            "studySlug": "swe-bench-verified",
            "statId": "median-minutes",
            "metric": "stat:median-minutes",
            "label": "Median worker time per attempt",
            "value": 9.6,
            "unit": "minutes",
            "display": "9.6 min",
            "n": 33,
            "context": "full pipeline on Claude Sonnet 5.5"
          },
          {
            "category": "cost",
            "studySlug": "swe-bench-verified",
            "statId": "cost-per-attempt",
            "metric": "stat:cost-per-attempt",
            "label": "Agent model cost per attempt (notional)",
            "value": 2.81,
            "unit": "usd",
            "display": "$2.81",
            "n": 33,
            "calculation": true,
            "context": "full pipeline on Claude Sonnet 5.5"
          },
          {
            "category": "cost",
            "studySlug": "cost-thought-experiments",
            "chartId": "cost-per-resolved-agent-vs-panel",
            "series": "Cost per resolved instance",
            "point": "Agent (notional)",
            "metric": "cost-per-resolved-agent-vs-panel",
            "label": "Recorded cost per resolved instance: Agent vs the public panel",
            "value": 3.706,
            "unit": "usd",
            "display": "$3.71",
            "n": 25,
            "calculation": true,
            "context": "full pipeline on Claude Sonnet 5.5 · notional"
          },
          {
            "category": "cost",
            "studySlug": "coding-calibration",
            "statId": "cost-latest",
            "metric": "stat:cost-latest",
            "label": "Notional cost, latest build, all 3 tasks",
            "value": 11.06,
            "unit": "usd",
            "display": "$11.06",
            "n": 3,
            "calculation": true,
            "context": "three real tasks, platform builds compared"
          }
        ]
      },
      {
        "entity": "gpt-5-2",
        "name": "GPT 5.2",
        "kind": "model",
        "vendor": "OpenAI",
        "measuredIn": 2,
        "studies": [
          "swe-bench-verified",
          "cost-thought-experiments"
        ],
        "factCount": 3,
        "best": [
          {
            "category": "quality",
            "studySlug": "swe-bench-verified",
            "chartId": "swebench-same-instance-leaderboard",
            "series": "Resolved rate",
            "point": "GPT 5.2 (high)",
            "metric": "swebench-same-instance-leaderboard",
            "label": "Resolved rate on the same 33 SWE-bench Verified instances",
            "value": 0.8485,
            "unit": "rate",
            "display": "85% (28/33)",
            "n": 33,
            "ci": [
              0.6908,
              0.9335
            ],
            "spanKind": "ci95",
            "context": "effort high · public mini-SWE-agent v2 run, same instances"
          },
          {
            "category": "cost",
            "studySlug": "cost-thought-experiments",
            "chartId": "cost-per-resolved-agent-vs-panel",
            "series": "Cost per resolved instance",
            "point": "GPT 5.2 (high)",
            "metric": "cost-per-resolved-agent-vs-panel",
            "label": "Recorded cost per resolved instance: Agent vs the public panel",
            "value": 0.628,
            "unit": "usd",
            "display": "$0.63",
            "n": 28,
            "context": "effort high · public mini-SWE-agent v2 run, same instances"
          }
        ]
      },
      {
        "entity": "gemini-3-flash",
        "name": "Gemini 3 Flash",
        "kind": "model",
        "vendor": "Google",
        "measuredIn": 2,
        "studies": [
          "swe-bench-verified",
          "cost-thought-experiments"
        ],
        "factCount": 3,
        "best": [
          {
            "category": "quality",
            "studySlug": "swe-bench-verified",
            "chartId": "swebench-same-instance-leaderboard",
            "series": "Resolved rate",
            "point": "Gemini 3 Flash (high)",
            "metric": "swebench-same-instance-leaderboard",
            "label": "Resolved rate on the same 33 SWE-bench Verified instances",
            "value": 0.8182,
            "unit": "rate",
            "display": "82% (27/33)",
            "n": 33,
            "ci": [
              0.6561,
              0.9139
            ],
            "spanKind": "ci95",
            "context": "effort high · public mini-SWE-agent v2 run, same instances"
          },
          {
            "category": "cost",
            "studySlug": "cost-thought-experiments",
            "chartId": "cost-per-resolved-agent-vs-panel",
            "series": "Cost per resolved instance",
            "point": "Gemini 3 Flash (high)",
            "metric": "cost-per-resolved-agent-vs-panel",
            "label": "Recorded cost per resolved instance: Agent vs the public panel",
            "value": 0.436,
            "unit": "usd",
            "display": "$0.44",
            "n": 27,
            "context": "effort high · public mini-SWE-agent v2 run, same instances"
          }
        ]
      },
      {
        "entity": "glm-5",
        "name": "GLM 5",
        "kind": "model",
        "vendor": "Z.ai",
        "measuredIn": 2,
        "studies": [
          "swe-bench-verified",
          "cost-thought-experiments"
        ],
        "factCount": 3,
        "best": [
          {
            "category": "quality",
            "studySlug": "swe-bench-verified",
            "chartId": "swebench-same-instance-leaderboard",
            "series": "Resolved rate",
            "point": "GLM 5 (high)",
            "metric": "swebench-same-instance-leaderboard",
            "label": "Resolved rate on the same 33 SWE-bench Verified instances",
            "value": 0.7879,
            "unit": "rate",
            "display": "79% (26/33)",
            "n": 33,
            "ci": [
              0.6225,
              0.8932
            ],
            "spanKind": "ci95",
            "context": "effort high · public mini-SWE-agent v2 run, same instances"
          },
          {
            "category": "cost",
            "studySlug": "cost-thought-experiments",
            "chartId": "cost-per-resolved-agent-vs-panel",
            "series": "Cost per resolved instance",
            "point": "GLM 5 (high)",
            "metric": "cost-per-resolved-agent-vs-panel",
            "label": "Recorded cost per resolved instance: Agent vs the public panel",
            "value": 0.667,
            "unit": "usd",
            "display": "$0.67",
            "n": 26,
            "context": "effort high · public mini-SWE-agent v2 run, same instances"
          }
        ]
      },
      {
        "entity": "claude-sonnet-4-5",
        "name": "Claude Sonnet 4.5",
        "kind": "model",
        "vendor": "Anthropic",
        "measuredIn": 2,
        "studies": [
          "swe-bench-verified",
          "cost-thought-experiments"
        ],
        "factCount": 3,
        "best": [
          {
            "category": "quality",
            "studySlug": "swe-bench-verified",
            "chartId": "swebench-same-instance-leaderboard",
            "series": "Resolved rate",
            "point": "Claude 4.5 Sonnet (high)",
            "metric": "swebench-same-instance-leaderboard",
            "label": "Resolved rate on the same 33 SWE-bench Verified instances",
            "value": 0.7576,
            "unit": "rate",
            "display": "76% (25/33)",
            "n": 33,
            "ci": [
              0.5898,
              0.8717
            ],
            "spanKind": "ci95",
            "context": "effort high · public mini-SWE-agent v2 run, same instances"
          },
          {
            "category": "cost",
            "studySlug": "cost-thought-experiments",
            "chartId": "cost-per-resolved-agent-vs-panel",
            "series": "Cost per resolved instance",
            "point": "Claude 4.5 Sonnet (high)",
            "metric": "cost-per-resolved-agent-vs-panel",
            "label": "Recorded cost per resolved instance: Agent vs the public panel",
            "value": 0.913,
            "unit": "usd",
            "display": "$0.91",
            "n": 25,
            "context": "effort high · public mini-SWE-agent v2 run, same instances"
          }
        ]
      },
      {
        "entity": "claude-opus-4-5",
        "name": "Claude Opus 4.5",
        "kind": "model",
        "vendor": "Anthropic",
        "measuredIn": 2,
        "studies": [
          "swe-bench-verified",
          "cost-thought-experiments"
        ],
        "factCount": 3,
        "best": [
          {
            "category": "quality",
            "studySlug": "swe-bench-verified",
            "chartId": "swebench-same-instance-leaderboard",
            "series": "Resolved rate",
            "point": "Claude 4.5 Opus (high)",
            "metric": "swebench-same-instance-leaderboard",
            "label": "Resolved rate on the same 33 SWE-bench Verified instances",
            "value": 0.7273,
            "unit": "rate",
            "display": "73% (24/33)",
            "n": 33,
            "ci": [
              0.5578,
              0.8493
            ],
            "spanKind": "ci95",
            "context": "effort high · public mini-SWE-agent v2 run, same instances"
          },
          {
            "category": "cost",
            "studySlug": "cost-thought-experiments",
            "chartId": "cost-per-resolved-agent-vs-panel",
            "series": "Cost per resolved instance",
            "point": "Claude 4.5 Opus (high)",
            "metric": "cost-per-resolved-agent-vs-panel",
            "label": "Recorded cost per resolved instance: Agent vs the public panel",
            "value": 1.184,
            "unit": "usd",
            "display": "$1.18",
            "n": 24,
            "context": "effort high · public mini-SWE-agent v2 run, same instances"
          }
        ]
      },
      {
        "entity": "claude-opus-4-6",
        "name": "Claude Opus 4.6",
        "kind": "model",
        "vendor": "Anthropic",
        "measuredIn": 2,
        "studies": [
          "swe-bench-verified",
          "cost-thought-experiments"
        ],
        "factCount": 3,
        "best": [
          {
            "category": "quality",
            "studySlug": "swe-bench-verified",
            "chartId": "swebench-same-instance-leaderboard",
            "series": "Resolved rate",
            "point": "Claude 4.6 Opus",
            "metric": "swebench-same-instance-leaderboard",
            "label": "Resolved rate on the same 33 SWE-bench Verified instances",
            "value": 0.697,
            "unit": "rate",
            "display": "70% (23/33)",
            "n": 33,
            "ci": [
              0.5266,
              0.8262
            ],
            "spanKind": "ci95",
            "context": "public mini-SWE-agent v2 run, same instances"
          },
          {
            "category": "cost",
            "studySlug": "cost-thought-experiments",
            "chartId": "cost-per-resolved-agent-vs-panel",
            "series": "Cost per resolved instance",
            "point": "Claude 4.6 Opus",
            "metric": "cost-per-resolved-agent-vs-panel",
            "label": "Recorded cost per resolved instance: Agent vs the public panel",
            "value": 0.875,
            "unit": "usd",
            "display": "$0.88",
            "n": 23,
            "context": "public mini-SWE-agent v2 run, same instances"
          }
        ]
      },
      {
        "entity": "deepseek-v3-2",
        "name": "DeepSeek V3.2",
        "kind": "model",
        "vendor": "DeepSeek",
        "measuredIn": 2,
        "studies": [
          "swe-bench-verified",
          "cost-thought-experiments"
        ],
        "factCount": 3,
        "best": [
          {
            "category": "quality",
            "studySlug": "swe-bench-verified",
            "chartId": "swebench-same-instance-leaderboard",
            "series": "Resolved rate",
            "point": "DeepSeek V3.2 (high)",
            "metric": "swebench-same-instance-leaderboard",
            "label": "Resolved rate on the same 33 SWE-bench Verified instances",
            "value": 0.7273,
            "unit": "rate",
            "display": "73% (24/33)",
            "n": 33,
            "ci": [
              0.5578,
              0.8493
            ],
            "spanKind": "ci95",
            "context": "effort high · public mini-SWE-agent v2 run, same instances"
          },
          {
            "category": "cost",
            "studySlug": "cost-thought-experiments",
            "chartId": "cost-per-resolved-agent-vs-panel",
            "series": "Cost per resolved instance",
            "point": "DeepSeek V3.2 (high)",
            "metric": "cost-per-resolved-agent-vs-panel",
            "label": "Recorded cost per resolved instance: Agent vs the public panel",
            "value": 0.637,
            "unit": "usd",
            "display": "$0.64",
            "n": 24,
            "context": "effort high · public mini-SWE-agent v2 run, same instances"
          }
        ]
      },
      {
        "entity": "minimax-m2-5",
        "name": "MiniMax M2.5",
        "kind": "model",
        "vendor": "MiniMax",
        "measuredIn": 2,
        "studies": [
          "swe-bench-verified",
          "cost-thought-experiments"
        ],
        "factCount": 3,
        "best": [
          {
            "category": "quality",
            "studySlug": "swe-bench-verified",
            "chartId": "swebench-same-instance-leaderboard",
            "series": "Resolved rate",
            "point": "MiniMax M2.5 (high)",
            "metric": "swebench-same-instance-leaderboard",
            "label": "Resolved rate on the same 33 SWE-bench Verified instances",
            "value": 0.697,
            "unit": "rate",
            "display": "70% (23/33)",
            "n": 33,
            "ci": [
              0.5266,
              0.8262
            ],
            "spanKind": "ci95",
            "context": "effort high · public mini-SWE-agent v2 run, same instances"
          },
          {
            "category": "cost",
            "studySlug": "cost-thought-experiments",
            "chartId": "cost-per-resolved-agent-vs-panel",
            "series": "Cost per resolved instance",
            "point": "MiniMax M2.5 (high)",
            "metric": "cost-per-resolved-agent-vs-panel",
            "label": "Recorded cost per resolved instance: Agent vs the public panel",
            "value": 0.107,
            "unit": "usd",
            "display": "$0.11",
            "n": 23,
            "context": "effort high · public mini-SWE-agent v2 run, same instances"
          }
        ]
      },
      {
        "entity": "kimi-k2-5",
        "name": "Kimi K2.5",
        "kind": "model",
        "vendor": "Moonshot AI",
        "measuredIn": 2,
        "studies": [
          "swe-bench-verified",
          "cost-thought-experiments"
        ],
        "factCount": 3,
        "best": [
          {
            "category": "quality",
            "studySlug": "swe-bench-verified",
            "chartId": "swebench-same-instance-leaderboard",
            "series": "Resolved rate",
            "point": "Kimi K2.5 (high)",
            "metric": "swebench-same-instance-leaderboard",
            "label": "Resolved rate on the same 33 SWE-bench Verified instances",
            "value": 0.697,
            "unit": "rate",
            "display": "70% (23/33)",
            "n": 33,
            "ci": [
              0.5266,
              0.8262
            ],
            "spanKind": "ci95",
            "context": "effort high · public mini-SWE-agent v2 run, same instances"
          },
          {
            "category": "cost",
            "studySlug": "cost-thought-experiments",
            "chartId": "cost-per-resolved-agent-vs-panel",
            "series": "Cost per resolved instance",
            "point": "Kimi K2.5 (high)",
            "metric": "cost-per-resolved-agent-vs-panel",
            "label": "Recorded cost per resolved instance: Agent vs the public panel",
            "value": 0.256,
            "unit": "usd",
            "display": "$0.26",
            "n": 23,
            "context": "effort high · public mini-SWE-agent v2 run, same instances"
          }
        ]
      },
      {
        "entity": "gpt-5-mini",
        "name": "GPT 5 mini",
        "kind": "model",
        "vendor": "OpenAI",
        "measuredIn": 2,
        "studies": [
          "swe-bench-verified",
          "cost-thought-experiments"
        ],
        "factCount": 3,
        "best": [
          {
            "category": "quality",
            "studySlug": "swe-bench-verified",
            "chartId": "swebench-same-instance-leaderboard",
            "series": "Resolved rate",
            "point": "GPT 5 mini",
            "metric": "swebench-same-instance-leaderboard",
            "label": "Resolved rate on the same 33 SWE-bench Verified instances",
            "value": 0.6364,
            "unit": "rate",
            "display": "64% (21/33)",
            "n": 33,
            "ci": [
              0.4662,
              0.7781
            ],
            "spanKind": "ci95",
            "context": "public mini-SWE-agent v2 run, same instances"
          },
          {
            "category": "cost",
            "studySlug": "cost-thought-experiments",
            "chartId": "cost-per-resolved-agent-vs-panel",
            "series": "Cost per resolved instance",
            "point": "GPT 5 mini",
            "metric": "cost-per-resolved-agent-vs-panel",
            "label": "Recorded cost per resolved instance: Agent vs the public panel",
            "value": 0.08,
            "unit": "usd",
            "display": "$0.080",
            "n": 21,
            "context": "public mini-SWE-agent v2 run, same instances"
          }
        ]
      },
      {
        "entity": "google-vertex",
        "name": "Google Vertex AI",
        "kind": "provider",
        "vendor": "Google",
        "measuredIn": 1,
        "studies": [
          "inference-provider-index"
        ],
        "factCount": 37,
        "best": [
          {
            "category": "cost",
            "studySlug": "inference-provider-index",
            "chartId": "provider-prices-claude-haiku-4-5",
            "series": "Input",
            "point": "Google Vertex",
            "metric": "provider-prices-claude-haiku-4-5/Input",
            "label": "Claude Haiku 4.5: price per million tokens by provider (Input)",
            "value": 1,
            "unit": "usd",
            "display": "$1.00",
            "context": "Claude Haiku 4.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
          }
        ]
      },
      {
        "entity": "anthropic",
        "name": "Anthropic",
        "kind": "provider",
        "vendor": "Anthropic",
        "measuredIn": 1,
        "studies": [
          "inference-provider-index"
        ],
        "factCount": 35,
        "best": [
          {
            "category": "cost",
            "studySlug": "inference-provider-index",
            "chartId": "gateway-vs-direct-claude-haiku-4-5",
            "series": "Input",
            "point": "Anthropic (first-party list price)",
            "metric": "gateway-vs-direct-claude-haiku-4-5/Input",
            "label": "Claude Haiku 4.5: OpenRouter vs Anthropic list price (Input)",
            "value": 1,
            "unit": "usd",
            "display": "$1.00",
            "context": "first-party list price · Claude Haiku 4.5 · list price, snapshot 2026-10-06"
          }
        ]
      },
      {
        "entity": "azure",
        "name": "Azure",
        "kind": "provider",
        "vendor": "Microsoft",
        "measuredIn": 1,
        "studies": [
          "inference-provider-index"
        ],
        "factCount": 33,
        "best": [
          {
            "category": "cost",
            "studySlug": "inference-provider-index",
            "chartId": "provider-prices-claude-haiku-4-5",
            "series": "Input",
            "point": "Azure",
            "metric": "provider-prices-claude-haiku-4-5/Input",
            "label": "Claude Haiku 4.5: price per million tokens by provider (Input)",
            "value": 1,
            "unit": "usd",
            "display": "$1.00",
            "context": "Claude Haiku 4.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
          }
        ]
      },
      {
        "entity": "amazon-bedrock",
        "name": "Amazon Bedrock",
        "kind": "provider",
        "vendor": "Amazon",
        "measuredIn": 1,
        "studies": [
          "inference-provider-index"
        ],
        "factCount": 23,
        "best": [
          {
            "category": "cost",
            "studySlug": "inference-provider-index",
            "chartId": "provider-prices-claude-haiku-4-5",
            "series": "Input",
            "point": "Amazon Bedrock",
            "metric": "provider-prices-claude-haiku-4-5/Input",
            "label": "Claude Haiku 4.5: price per million tokens by provider (Input)",
            "value": 1,
            "unit": "usd",
            "display": "$1.00",
            "context": "Claude Haiku 4.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
          }
        ]
      },
      {
        "entity": "openrouter",
        "name": "OpenRouter",
        "kind": "provider",
        "vendor": "OpenRouter",
        "measuredIn": 1,
        "studies": [
          "inference-provider-index"
        ],
        "factCount": 20,
        "best": [
          {
            "category": "cost",
            "studySlug": "inference-provider-index",
            "chartId": "gateway-vs-direct-claude-haiku-4-5",
            "series": "Input",
            "point": "OpenRouter",
            "metric": "gateway-vs-direct-claude-haiku-4-5/Input",
            "label": "Claude Haiku 4.5: OpenRouter vs Anthropic list price (Input)",
            "value": 1,
            "unit": "usd",
            "display": "$1.00",
            "context": "Claude Haiku 4.5 · list price, snapshot 2026-10-06"
          }
        ]
      },
      {
        "entity": "parasail",
        "name": "Parasail",
        "kind": "provider",
        "vendor": "Parasail",
        "measuredIn": 1,
        "studies": [
          "inference-provider-index"
        ],
        "factCount": 20,
        "best": [
          {
            "category": "cost",
            "studySlug": "inference-provider-index",
            "chartId": "provider-prices-gpt-oss-120b",
            "series": "Input",
            "point": "Parasail (fp4)",
            "metric": "provider-prices-gpt-oss-120b/Input",
            "label": "gpt-oss-120b: price per million tokens by provider (Input)",
            "value": 0.1,
            "unit": "usd",
            "display": "$0.10",
            "context": "fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
          }
        ]
      },
      {
        "entity": "google-ai-studio",
        "name": "Google AI Studio",
        "kind": "provider",
        "vendor": "Google",
        "measuredIn": 1,
        "studies": [
          "inference-provider-index"
        ],
        "factCount": 16,
        "best": [
          {
            "category": "cost",
            "studySlug": "inference-provider-index",
            "chartId": "gateway-vs-direct-gemini-3-8-flash",
            "series": "Input",
            "point": "Google AI Studio (first-party list price)",
            "metric": "gateway-vs-direct-gemini-3-8-flash/Input",
            "label": "Gemini 3.8 Flash: OpenRouter vs Google list price (Input)",
            "value": 0.75,
            "unit": "usd",
            "display": "$0.75",
            "context": "first-party list price · Gemini 3.8 Flash · list price, snapshot 2026-10-06"
          }
        ]
      },
      {
        "entity": "deepinfra",
        "name": "DeepInfra",
        "kind": "provider",
        "vendor": "DeepInfra",
        "measuredIn": 1,
        "studies": [
          "inference-provider-index"
        ],
        "factCount": 16,
        "best": [
          {
            "category": "cost",
            "studySlug": "inference-provider-index",
            "chartId": "provider-prices-gpt-oss-120b",
            "series": "Input",
            "point": "DeepInfra (bf16)",
            "metric": "provider-prices-gpt-oss-120b/Input",
            "label": "gpt-oss-120b: price per million tokens by provider (Input)",
            "value": 0.037,
            "unit": "usd",
            "display": "$0.037",
            "context": "bf16 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
          }
        ]
      },
      {
        "entity": "claude-platform-on-aws",
        "name": "Claude Platform on AWS",
        "kind": "provider",
        "vendor": "Anthropic",
        "measuredIn": 1,
        "studies": [
          "inference-provider-index"
        ],
        "factCount": 15,
        "best": [
          {
            "category": "cost",
            "studySlug": "inference-provider-index",
            "chartId": "provider-prices-claude-sonnet-5",
            "series": "Input",
            "point": "Claude Platform on AWS",
            "metric": "provider-prices-claude-sonnet-5/Input",
            "label": "Claude Sonnet 5: price per million tokens by provider (Input)",
            "value": 2,
            "unit": "usd",
            "display": "$2.00",
            "context": "Claude Sonnet 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"
          }
        ]
      },
      {
        "entity": "novita",
        "name": "Novita AI",
        "kind": "provider",
        "vendor": "Novita AI",
        "measuredIn": 1,
        "studies": [
          "inference-provider-index"
        ],
        "factCount": 15,
        "best": [
          {
            "category": "cost",
            "studySlug": "inference-provider-index",
            "chartId": "provider-prices-gpt-oss-120b",
            "series": "Input",
            "point": "Novita (fp4)",
            "metric": "provider-prices-gpt-oss-120b/Input",
            "label": "gpt-oss-120b: price per million tokens by provider (Input)",
            "value": 0.05,
            "unit": "usd",
            "display": "$0.050",
            "context": "fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
          }
        ]
      },
      {
        "entity": "openai",
        "name": "OpenAI",
        "kind": "provider",
        "vendor": "OpenAI",
        "measuredIn": 1,
        "studies": [
          "inference-provider-index"
        ],
        "factCount": 14,
        "best": [
          {
            "category": "cost",
            "studySlug": "inference-provider-index",
            "chartId": "gateway-vs-direct-gpt-6-luna",
            "series": "Input",
            "point": "OpenAI (first-party list price)",
            "metric": "gateway-vs-direct-gpt-6-luna/Input",
            "label": "GPT-6 Luna: OpenRouter vs OpenAI list price (Input)",
            "value": 0.1,
            "unit": "usd",
            "display": "$0.10",
            "context": "first-party list price · GPT-6 Luna · list price, snapshot 2026-10-06"
          }
        ]
      },
      {
        "entity": "gpt-6-1-sol-openai-api",
        "name": "GPT-6.1 Sol (OpenAI API)",
        "kind": "model",
        "vendor": "OpenAI",
        "measuredIn": 1,
        "studies": [
          "cli-model-latency-tokens"
        ],
        "factCount": 13,
        "best": [
          {
            "category": "speed",
            "studySlug": "cli-model-latency-tokens",
            "chartId": "cli-vs-api-exact-reply-latency",
            "series": "Total time",
            "point": "OpenAI API · GPT-6.1 Sol · high",
            "metric": "cli-vs-api-exact-reply-latency/Total time",
            "label": "CLI vs API: time for a one-line answer (Total time)",
            "value": 1.52,
            "unit": "seconds",
            "display": "1.52 s",
            "n": 5,
            "range": [
              1.35,
              2.23
            ],
            "spanKind": "minmax",
            "context": "OpenAI API · effort high · fixed exact reply, 5 runs"
          }
        ]
      },
      {
        "entity": "siliconflow",
        "name": "SiliconFlow",
        "kind": "provider",
        "vendor": "SiliconFlow",
        "measuredIn": 1,
        "studies": [
          "inference-provider-index"
        ],
        "factCount": 12,
        "best": [
          {
            "category": "cost",
            "studySlug": "inference-provider-index",
            "chartId": "provider-prices-gpt-oss-120b",
            "series": "Input",
            "point": "SiliconFlow (fp8)",
            "metric": "provider-prices-gpt-oss-120b/Input",
            "label": "gpt-oss-120b: price per million tokens by provider (Input)",
            "value": 0.15,
            "unit": "usd",
            "display": "$0.15",
            "context": "fp8 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
          }
        ]
      },
      {
        "entity": "cloudflare",
        "name": "Cloudflare Workers AI",
        "kind": "provider",
        "vendor": "Cloudflare",
        "measuredIn": 1,
        "studies": [
          "inference-provider-index"
        ],
        "factCount": 11,
        "best": [
          {
            "category": "cost",
            "studySlug": "inference-provider-index",
            "chartId": "provider-prices-llama-3-3-70b-instruct",
            "series": "Input",
            "point": "Cloudflare (fp8)",
            "metric": "provider-prices-llama-3-3-70b-instruct/Input",
            "label": "Llama 3.3 70B Instruct: price per million tokens by provider (Input)",
            "value": 0.293,
            "unit": "usd",
            "display": "$0.29",
            "context": "fp8 · Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06"
          }
        ]
      },
      {
        "entity": "together",
        "name": "Together AI",
        "kind": "provider",
        "vendor": "Together AI",
        "measuredIn": 1,
        "studies": [
          "inference-provider-index"
        ],
        "factCount": 10,
        "best": [
          {
            "category": "cost",
            "studySlug": "inference-provider-index",
            "chartId": "provider-prices-gpt-oss-120b",
            "series": "Input",
            "point": "Together",
            "metric": "provider-prices-gpt-oss-120b/Input",
            "label": "gpt-oss-120b: price per million tokens by provider (Input)",
            "value": 0.15,
            "unit": "usd",
            "display": "$0.15",
            "context": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
          }
        ]
      },
      {
        "entity": "baseten",
        "name": "Baseten",
        "kind": "provider",
        "vendor": "Baseten",
        "measuredIn": 1,
        "studies": [
          "inference-provider-index"
        ],
        "factCount": 9,
        "best": [
          {
            "category": "cost",
            "studySlug": "inference-provider-index",
            "chartId": "provider-prices-gpt-oss-120b",
            "series": "Input",
            "point": "BaseTen (fp4)",
            "metric": "provider-prices-gpt-oss-120b/Input",
            "label": "gpt-oss-120b: price per million tokens by provider (Input)",
            "value": 0.1,
            "unit": "usd",
            "display": "$0.10",
            "context": "fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
          }
        ]
      },
      {
        "entity": "deterministic-routing-policy",
        "name": "Deterministic routing policy",
        "kind": "router",
        "vendor": "Agent",
        "measuredIn": 1,
        "studies": [
          "routing-overhead"
        ],
        "factCount": 7,
        "best": [
          {
            "category": "quality",
            "studySlug": "routing-overhead",
            "chartId": "router-overhead-completed",
            "series": "Completed",
            "point": "Deterministic routing policy (Agent, in process)",
            "metric": "router-overhead-completed",
            "label": "Routing calls that returned a decision",
            "value": 1,
            "unit": "rate",
            "display": "100% (20000/20000)",
            "n": 20000,
            "ci": [
              0.9998,
              1
            ],
            "spanKind": "ci95",
            "context": "Agent · in process · routing overhead per decision"
          },
          {
            "category": "speed",
            "studySlug": "routing-overhead",
            "chartId": "router-overhead-decision-latency",
            "series": "Decision time",
            "point": "Deterministic routing policy (Agent, in process)",
            "metric": "router-overhead-decision-latency",
            "label": "Time to make one routing decision",
            "value": 0.00142,
            "unit": "ms",
            "display": "1.42 µs",
            "n": 20000,
            "range": [
              0.00142,
              0.00233
            ],
            "spanKind": "p50-p95",
            "context": "Agent · in process · routing overhead per decision"
          },
          {
            "category": "cost",
            "studySlug": "routing-overhead",
            "chartId": "router-overhead-cost-reported",
            "series": "Cost per 1,000 decisions",
            "point": "Deterministic routing policy (Agent, in process)",
            "metric": "router-overhead-cost-reported",
            "label": "Cost per 1,000 routing decisions: no model call vs provider-reported",
            "value": 0,
            "unit": "usd",
            "display": "$0.00",
            "n": 20000,
            "context": "Agent · in process · routing overhead per decision"
          }
        ]
      },
      {
        "entity": "groq",
        "name": "Groq",
        "kind": "provider",
        "vendor": "Groq",
        "measuredIn": 1,
        "studies": [
          "inference-provider-index"
        ],
        "factCount": 6,
        "best": [
          {
            "category": "cost",
            "studySlug": "inference-provider-index",
            "chartId": "provider-prices-gpt-oss-120b",
            "series": "Input",
            "point": "Groq",
            "metric": "provider-prices-gpt-oss-120b/Input",
            "label": "gpt-oss-120b: price per million tokens by provider (Input)",
            "value": 0.15,
            "unit": "usd",
            "display": "$0.15",
            "context": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
          }
        ]
      },
      {
        "entity": "fireworks",
        "name": "Fireworks AI",
        "kind": "provider",
        "vendor": "Fireworks AI",
        "measuredIn": 1,
        "studies": [
          "inference-provider-index"
        ],
        "factCount": 6,
        "best": [
          {
            "category": "cost",
            "studySlug": "inference-provider-index",
            "chartId": "provider-prices-kimi-k3",
            "series": "Input",
            "point": "Fireworks",
            "metric": "provider-prices-kimi-k3/Input",
            "label": "Kimi K3: price per million tokens by provider (Input)",
            "value": 3,
            "unit": "usd",
            "display": "$3.00",
            "context": "Kimi K3 · reported by OpenRouter’s public API, snapshot 2026-10-06"
          }
        ]
      },
      {
        "entity": "gpt-6-luna-openai-api",
        "name": "GPT-6 Luna (OpenAI API)",
        "kind": "model",
        "vendor": "OpenAI",
        "measuredIn": 1,
        "studies": [
          "cli-model-latency-tokens"
        ],
        "factCount": 5,
        "best": [
          {
            "category": "speed",
            "studySlug": "cli-model-latency-tokens",
            "chartId": "cli-vs-api-exact-reply-latency",
            "series": "Total time",
            "point": "OpenAI API · GPT-6 Luna · none",
            "metric": "cli-vs-api-exact-reply-latency/Total time",
            "label": "CLI vs API: time for a one-line answer (Total time)",
            "value": 0.97,
            "unit": "seconds",
            "display": "0.97 s",
            "n": 5,
            "range": [
              0.65,
              1.5
            ],
            "spanKind": "minmax",
            "context": "OpenAI API · effort none · fixed exact reply, 5 runs"
          }
        ]
      },
      {
        "entity": "sambanova",
        "name": "SambaNova",
        "kind": "provider",
        "vendor": "SambaNova",
        "measuredIn": 1,
        "studies": [
          "inference-provider-index"
        ],
        "factCount": 4,
        "best": [
          {
            "category": "cost",
            "studySlug": "inference-provider-index",
            "chartId": "provider-prices-gpt-oss-120b",
            "series": "Input",
            "point": "SambaNova",
            "metric": "provider-prices-gpt-oss-120b/Input",
            "label": "gpt-oss-120b: price per million tokens by provider (Input)",
            "value": 0.14,
            "unit": "usd",
            "display": "$0.14",
            "context": "gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
          }
        ]
      },
      {
        "entity": "nebius",
        "name": "Nebius",
        "kind": "provider",
        "vendor": "Nebius",
        "measuredIn": 1,
        "studies": [
          "inference-provider-index"
        ],
        "factCount": 4,
        "best": [
          {
            "category": "cost",
            "studySlug": "inference-provider-index",
            "chartId": "provider-prices-gpt-oss-120b",
            "series": "Input",
            "point": "Nebius (fp4)",
            "metric": "provider-prices-gpt-oss-120b/Input",
            "label": "gpt-oss-120b: price per million tokens by provider (Input)",
            "value": 0.15,
            "unit": "usd",
            "display": "$0.15",
            "context": "fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
          }
        ]
      },
      {
        "entity": "cerebras",
        "name": "Cerebras",
        "kind": "provider",
        "vendor": "Cerebras",
        "measuredIn": 1,
        "studies": [
          "inference-provider-index"
        ],
        "factCount": 3,
        "best": [
          {
            "category": "cost",
            "studySlug": "inference-provider-index",
            "chartId": "provider-prices-gpt-oss-120b",
            "series": "Input",
            "point": "Cerebras (fp16)",
            "metric": "provider-prices-gpt-oss-120b/Input",
            "label": "gpt-oss-120b: price per million tokens by provider (Input)",
            "value": 0.35,
            "unit": "usd",
            "display": "$0.35",
            "context": "fp16 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"
          }
        ]
      },
      {
        "entity": "claude-opus-5",
        "name": "Claude Opus 5",
        "kind": "model",
        "vendor": "Anthropic",
        "measuredIn": 1,
        "studies": [
          "blind-review-head-to-head"
        ],
        "factCount": 1,
        "best": [
          {
            "category": "quality",
            "studySlug": "blind-review-head-to-head",
            "chartId": "blind-review-critic-agreement",
            "series": "Critic model",
            "point": "Claude Opus 5 (Anthropic)",
            "metric": "blind-review-critic-agreement",
            "label": "Does the judge’s model family matter?",
            "value": 0.7,
            "unit": "rate",
            "display": "70% (28/40)",
            "n": 40,
            "ci": [
              0.5457,
              0.8193
            ],
            "spanKind": "ci95",
            "context": "blind review panel"
          }
        ]
      },
      {
        "entity": "claude-fable-5",
        "name": "Claude Fable 5",
        "kind": "model",
        "vendor": "Anthropic",
        "measuredIn": 1,
        "studies": [
          "blind-review-head-to-head"
        ],
        "factCount": 1,
        "best": [
          {
            "category": "quality",
            "studySlug": "blind-review-head-to-head",
            "chartId": "blind-review-critic-agreement",
            "series": "Critic model",
            "point": "Claude Fable 5 (Anthropic)",
            "metric": "blind-review-critic-agreement",
            "label": "Does the judge’s model family matter?",
            "value": 0.65,
            "unit": "rate",
            "display": "65% (26/40)",
            "n": 40,
            "ci": [
              0.4951,
              0.7787
            ],
            "spanKind": "ci95",
            "context": "blind review panel"
          }
        ]
      },
      {
        "entity": "gpt-5-5",
        "name": "GPT 5.5",
        "kind": "model",
        "vendor": "OpenAI",
        "measuredIn": 1,
        "studies": [
          "blind-review-head-to-head"
        ],
        "factCount": 1,
        "best": [
          {
            "category": "quality",
            "studySlug": "blind-review-head-to-head",
            "chartId": "blind-review-critic-agreement",
            "series": "Critic model",
            "point": "GPT 5.5 (OpenAI)",
            "metric": "blind-review-critic-agreement",
            "label": "Does the judge’s model family matter?",
            "value": 1,
            "unit": "rate",
            "display": "100% (2/2)",
            "n": 2,
            "ci": [
              0.3424,
              1
            ],
            "spanKind": "ci95",
            "context": "blind review panel"
          }
        ]
      }
    ]
  }
}
