{"$k":["slug","name","vendor","kind","description","aliases","facts"],"$r":[["claude-sonnet-5-5","Claude Sonnet 5.5","Anthropic","model","Anthropic’s mid-tier Claude model. Agent’s default coding model; measured here through Claude Code at low, medium, high and default effort, and as an LLM router.",["Claude Sonnet 5.5","Sonnet 5.5"],{"$k":["studySlug","chartId","series","point","metric","label","value","unit","display","n","ci","spanKind","context","calculation","range","polarity","statId"],"$r":[["swe-bench-opus-vs-sonnet","swebench-opus-sonnet-resolved","Resolved","Claude Sonnet 5.5 (Agent, older builds)","swebench-opus-sonnet-resolved","Resolved on the same 3 SWE-bench Verified instances (interim)",0.3333,"rate","33% (1/3)",3,[0.0615,0.7923],"ci95","Agent · older builds · SWE-bench Verified, interim paired probe","\u0001","\u0001","\u0001","\u0001"],["swe-bench-opus-vs-sonnet","swebench-opus-sonnet-cost-per-attempt","List-price cost per attempt","Claude Sonnet 5.5 (Agent, older builds)","swebench-opus-sonnet-cost-per-attempt","List-price cost per attempt (calculation)",2.88,"usd","$2.88",3,"\u0001","\u0001","Agent · older builds · SWE-bench Verified, interim paired probe",true,"\u0001","\u0001","\u0001"],["swe-bench-opus-vs-sonnet","swebench-opus-sonnet-minutes","Worker minutes per attempt","Claude Sonnet 5.5 (Agent, older builds)","swebench-opus-sonnet-minutes","Worker time per attempt",9.37,"minutes","9.4 min",3,"\u0001","minmax","Agent · older builds · SWE-bench Verified, interim paired probe","\u0001",[4.74,15],"\u0001","\u0001"],["model-head-to-head","h2h-pass-rate","Pass rate","Claude Sonnet 5.5 · Claude Code","h2h-pass-rate","Pass rate on five validated tasks",0.8,"rate","80% (12/15)",15,[0.5481,0.9295],"ci95","Claude Code · five short validated tasks","\u0001","\u0001","\u0001","\u0001"],["model-head-to-head","h2h-total-latency","Total time per call","Claude Sonnet 5.5 · Claude Code","h2h-total-latency","Total time per call",2.31,"seconds","2.31 s",15,"\u0001","minmax","Claude Code · five short validated tasks","\u0001",[2.17,7.73],"\u0001","\u0001"],["model-head-to-head","h2h-first-useful-latency","Time to first useful output","Claude Sonnet 5.5 · Claude Code","h2h-first-useful-latency","Time to first useful output",1.56,"seconds","1.56 s",15,"\u0001","minmax","Claude Code · five short validated tasks","\u0001",[0.99,6.39],"\u0001","\u0001"],["model-head-to-head","h2h-input-tokens","Cache read","Claude Sonnet 5.5 · Claude Code","h2h-input-tokens/Cache read","Input tokens per call: what the CLI sends (Cache read)",1401,"tokens","1,401",15,"\u0001","\u0001","Claude Code · five short validated tasks","\u0001","\u0001","\u0001","\u0001"],["model-head-to-head","h2h-input-tokens","Other input","Claude Sonnet 5.5 · Claude Code","h2h-input-tokens/Other input","Input tokens per call: what the CLI sends (Other input)",685,"tokens","685",15,"\u0001","\u0001","Claude Code · five short validated tasks","\u0001","\u0001","\u0001","\u0001"],["model-head-to-head","h2h-output-tokens","Output tokens","Claude Sonnet 5.5 · Claude Code","h2h-output-tokens/Output tokens","Output tokens per call (Output tokens)",107,"tokens","107",15,"\u0001","\u0001","Claude Code · five short validated tasks","\u0001","\u0001","\u0001","\u0001"],["model-head-to-head","h2h-list-price-per-call","Cost per call","Claude Sonnet 5.5 · Claude Code","h2h-list-price-per-call","List-price cost per call (calculation)",0.0036,"usd","$0.0036",15,"\u0001","minmax","Claude Code · five short validated tasks",true,[0.00342,0.01021],"\u0001","\u0001"],["model-head-to-head","h2h-cost-per-pass","Cost per pass","Claude Sonnet 5.5 · Claude Code","h2h-cost-per-pass","List-price cost per passing answer (calculation)",0.00624,"usd","$0.0062",15,"\u0001","\u0001","Claude Code · five short validated tasks",true,"\u0001","\u0001","\u0001"],["hard-model-head-to-head","hard-h2h-pass-rate","Strict pass","Claude Sonnet 5.5 · Claude Code","hard-h2h-pass-rate/Strict pass","Pass rate on eight hard tasks (Strict pass)",1,"rate","100% (24/24)",24,[0.862,1],"ci95","Claude Code · eight hard validated tasks","\u0001","\u0001","\u0001","\u0001"],["hard-model-head-to-head","hard-h2h-pass-rate","Lenient (format misses counted)","Claude Sonnet 5.5 · Claude Code","hard-h2h-pass-rate/Lenient (format misses counted)","Pass rate on eight hard tasks (Lenient (format misses counted))",1,"rate","100% (24/24)",24,[0.862,1],"ci95","Claude Code · eight hard validated tasks","\u0001","\u0001","\u0001","\u0001"],["hard-model-head-to-head","hard-h2h-total-latency","Total time per call on hard tasks (separate batches)","Claude Sonnet 5.5 · Claude Code","hard-h2h-total-latency","Total time per call on hard tasks (separate batches)",7.75,"seconds","7.75 s",24,"\u0001","minmax","Claude Code · eight hard validated tasks","\u0001",[2.26,34.79],"\u0001","\u0001"],["hard-model-head-to-head","hard-h2h-first-useful-latency","Time to first useful output on hard tasks","Claude Sonnet 5.5 · Claude Code","hard-h2h-first-useful-latency","Time to first useful output on hard tasks",5.95,"seconds","5.95 s",24,"\u0001","minmax","Claude Code · eight hard validated tasks","\u0001",[0.86,30.57],"\u0001","\u0001"],["hard-model-head-to-head","hard-h2h-output-tokens","Output tokens","Claude Sonnet 5.5 · Claude Code","hard-h2h-output-tokens/Output tokens","Output tokens per call on hard tasks (Output tokens)",1050,"tokens","1,050",24,"\u0001","\u0001","Claude Code · eight hard validated tasks","\u0001","\u0001","\u0001","\u0001"],["hard-model-head-to-head","hard-h2h-cost-per-pass","Cost per strict pass","Claude Sonnet 5.5 · Claude Code","hard-h2h-cost-per-pass","List-price cost per strict pass on hard tasks (calculation)",0.01435,"usd","$0.014",24,"\u0001","\u0001","Claude Code · eight hard validated tasks",true,"\u0001","\u0001","\u0001"],["coding-agents-head-to-head","coding-agents-pass-rate","Passed every hidden check","Claude Sonnet 5.5 · Claude Code","coding-agents-pass-rate","Coding sessions that passed every hidden check",1,"rate","100% (12/12)",12,[0.7575,1],"ci95","Claude Code · six small repository tasks with hidden tests","\u0001","\u0001","\u0001","\u0001"],["coding-agents-head-to-head","coding-agents-wall-time","Wall time per session","Claude Sonnet 5.5 · Claude Code","coding-agents-wall-time","Time per coding session",23.1,"seconds","23.1 s",12,"\u0001","minmax","Claude Code · six small repository tasks with hidden tests","\u0001",[18.7,44.5],"\u0001","\u0001"],["coding-agents-head-to-head","coding-agents-tool-calls","Tool calls per session","Claude Sonnet 5.5 · Claude Code","coding-agents-tool-calls","Tool calls per coding session",7.5,"calls","7.5",12,"\u0001","minmax","Claude Code · six small repository tasks with hidden tests","\u0001",[3,14],"\u0001","\u0001"],["coding-agents-head-to-head","coding-agents-cost-per-pass","List-price cost per pass","Claude Sonnet 5.5 · Claude Code","coding-agents-cost-per-pass","List-price cost per passing coding session (calculation)",0.085,"usd","$0.085",12,"\u0001","\u0001","Claude Code · six small repository tasks with hidden tests",true,"\u0001","\u0001","\u0001"],["effort-ladder","effort-ladder-pass-rate","Strict pass","Claude Sonnet 5.5 (low) · Claude Code","effort-ladder-pass-rate","Strict pass rate by effort on eight hard tasks",1,"rate","100% (16/16)",16,[0.8064,1],"ci95","Claude Code · effort low · eight hard validated tasks, effort ladder","\u0001","\u0001","\u0001","\u0001"],["effort-ladder","effort-ladder-pass-rate","Strict pass","Claude Sonnet 5.5 (medium) · Claude Code","effort-ladder-pass-rate","Strict pass rate by effort on eight hard tasks",1,"rate","100% (16/16)",16,[0.8064,1],"ci95","Claude Code · effort medium · eight hard validated tasks, effort ladder","\u0001","\u0001","\u0001","\u0001"],["effort-ladder","effort-ladder-pass-rate","Strict pass","Claude Sonnet 5.5 (high) · Claude Code","effort-ladder-pass-rate","Strict pass rate by effort on eight hard tasks",1,"rate","100% (16/16)",16,[0.8064,1],"ci95","Claude Code · effort high · eight hard validated tasks, effort ladder","\u0001","\u0001","\u0001","\u0001"],["effort-ladder","effort-ladder-pass-rate","Strict pass","Claude Sonnet 5.5 · Claude Code","effort-ladder-pass-rate","Strict pass rate by effort on eight hard tasks",1,"rate","100% (16/16)",16,[0.8064,1],"ci95","Claude Code · eight hard validated tasks, effort ladder","\u0001","\u0001","\u0001","\u0001"],["effort-ladder","effort-ladder-total-latency","Total time per call by effort on hard tasks","Claude Sonnet 5.5 (low) · Claude Code","effort-ladder-total-latency","Total time per call by effort on hard tasks",5.82,"seconds","5.82 s",16,"\u0001","minmax","Claude Code · effort low · eight hard validated tasks, effort ladder","\u0001",[2.78,19.96],"\u0001","\u0001"],["effort-ladder","effort-ladder-total-latency","Total time per call by effort on hard tasks","Claude Sonnet 5.5 (medium) · Claude Code","effort-ladder-total-latency","Total time per call by effort on hard tasks",7.63,"seconds","7.63 s",16,"\u0001","minmax","Claude Code · effort medium · eight hard validated tasks, effort ladder","\u0001",[2.71,24.01],"\u0001","\u0001"],["effort-ladder","effort-ladder-total-latency","Total time per call by effort on hard tasks","Claude Sonnet 5.5 (high) · Claude Code","effort-ladder-total-latency","Total time per call by effort on hard tasks",8.81,"seconds","8.81 s",16,"\u0001","minmax","Claude Code · effort high · eight hard validated tasks, effort ladder","\u0001",[2.93,35.81],"\u0001","\u0001"],["effort-ladder","effort-ladder-total-latency","Total time per call by effort on hard tasks","Claude Sonnet 5.5 · Claude Code","effort-ladder-total-latency","Total time per call by effort on hard tasks",7.97,"seconds","7.97 s",16,"\u0001","minmax","Claude Code · eight hard validated tasks, effort ladder","\u0001",[2.26,21.61],"\u0001","\u0001"],["effort-ladder","effort-ladder-output-tokens","Output tokens","Claude Sonnet 5.5 (low) · Claude Code","effort-ladder-output-tokens/Output tokens","Output tokens per call by effort on hard tasks (Output tokens)",667,"tokens","667",16,"\u0001","\u0001","Claude Code · effort low · eight hard validated tasks, effort ladder","\u0001","\u0001","\u0001","\u0001"],["effort-ladder","effort-ladder-output-tokens","Output tokens","Claude Sonnet 5.5 (medium) · Claude Code","effort-ladder-output-tokens/Output tokens","Output tokens per call by effort on hard tasks (Output tokens)",770,"tokens","770",16,"\u0001","\u0001","Claude Code · effort medium · eight hard validated tasks, effort ladder","\u0001","\u0001","\u0001","\u0001"],["effort-ladder","effort-ladder-output-tokens","Output tokens","Claude Sonnet 5.5 (high) · Claude Code","effort-ladder-output-tokens/Output tokens","Output tokens per call by effort on hard tasks (Output tokens)",1192,"tokens","1,192",16,"\u0001","\u0001","Claude Code · effort high · eight hard validated tasks, effort ladder","\u0001","\u0001","\u0001","\u0001"],["effort-ladder","effort-ladder-output-tokens","Output tokens","Claude Sonnet 5.5 · Claude Code","effort-ladder-output-tokens/Output tokens","Output tokens per call by effort on hard tasks (Output tokens)",1054,"tokens","1,054",16,"\u0001","\u0001","Claude Code · eight hard validated tasks, effort ladder","\u0001","\u0001","\u0001","\u0001"],["effort-ladder","effort-ladder-cost-per-pass","Cost per strict pass","Claude Sonnet 5.5 (low) · Claude Code","effort-ladder-cost-per-pass","List-price cost per strict pass by effort (calculation)",0.01219,"usd","$0.012",16,"\u0001","\u0001","Claude Code · effort low · eight hard validated tasks, effort ladder",true,"\u0001","\u0001","\u0001"],["effort-ladder","effort-ladder-cost-per-pass","Cost per strict pass","Claude Sonnet 5.5 (medium) · Claude Code","effort-ladder-cost-per-pass","List-price cost per strict pass by effort (calculation)",0.01352,"usd","$0.014",16,"\u0001","\u0001","Claude Code · effort medium · eight hard validated tasks, effort ladder",true,"\u0001","\u0001","\u0001"],["effort-ladder","effort-ladder-cost-per-pass","Cost per strict pass","Claude Sonnet 5.5 (high) · Claude Code","effort-ladder-cost-per-pass","List-price cost per strict pass by effort (calculation)",0.01671,"usd","$0.017",16,"\u0001","\u0001","Claude Code · effort high · eight hard validated tasks, effort ladder",true,"\u0001","\u0001","\u0001"],["effort-ladder","effort-ladder-cost-per-pass","Cost per strict pass","Claude Sonnet 5.5 · Claude Code","effort-ladder-cost-per-pass","List-price cost per strict pass by effort (calculation)",0.01398,"usd","$0.014",16,"\u0001","\u0001","Claude Code · eight hard validated tasks, effort ladder",true,"\u0001","\u0001","\u0001"],["caching-consistency","caching-cost-with-without","With the cache, as recorded","Claude Sonnet 5.5 · Claude Code","caching-cost-with-without/With the cache, as recorded","List-price cost of 5-question sessions with and without the cache (calculation) (With the cache, as recorded)",0.135003,"usd","$0.14",15,"\u0001","\u0001","Claude Code · calculation: 5-turn cached sessions over a fixed ledger",true,"\u0001","\u0001","\u0001"],["caching-consistency","caching-cost-with-without","Without a cache: every input token at the input price","Claude Sonnet 5.5 · Claude Code","caching-cost-with-without/Without a cache: every input token at the input price","List-price cost of 5-question sessions with and without the cache (calculation) (Without a cache: every input token at the input price)",0.269788,"usd","$0.27",15,"\u0001","\u0001","Claude Code · calculation: 5-turn cached sessions over a fixed ledger",true,"\u0001","\u0001","\u0001"],["caching-consistency","caching-latency-first-vs-later","Turn 1 (writes the ledger to the cache)","Claude Sonnet 5.5 · Claude Code","caching-latency-first-vs-later/Turn 1 (writes the ledger to the cache)","Time per turn: first turn vs later turns in a cached session (Turn 1 (writes the ledger to the cache))",1.64,"seconds","1.64 s",3,"\u0001","minmax","Claude Code · 5-turn cached sessions over a fixed ledger","\u0001",[1.58,1.79],"\u0001","\u0001"],["caching-consistency","caching-latency-first-vs-later","Turns 2-5 (read the ledger from the cache)","Claude Sonnet 5.5 · Claude Code","caching-latency-first-vs-later/Turns 2-5 (read the ledger from the cache)","Time per turn: first turn vs later turns in a cached session (Turns 2-5 (read the ledger from the cache))",1.61,"seconds","1.61 s",12,"\u0001","minmax","Claude Code · 5-turn cached sessions over a fixed ledger","\u0001",[1.35,5.63],"\u0001","\u0001"],["caching-consistency","consistency-pass-rate","Exact number","Claude Sonnet 5.5 · Claude Code","consistency-pass-rate/Exact number","Same prompt, 10 times: strict pass rate (Exact number)",1,"rate","100% (10/10)",10,[0.7225,1],"ci95","Claude Code · same prompt repeated 10 times","\u0001","\u0001","\u0001","\u0001"],["caching-consistency","consistency-pass-rate","JSON object","Claude Sonnet 5.5 · Claude Code","consistency-pass-rate/JSON object","Same prompt, 10 times: strict pass rate (JSON object)",1,"rate","100% (10/10)",10,[0.7225,1],"ci95","Claude Code · same prompt repeated 10 times","\u0001","\u0001","\u0001","\u0001"],["caching-consistency","consistency-pass-rate","Code fix","Claude Sonnet 5.5 · Claude Code","consistency-pass-rate/Code fix","Same prompt, 10 times: strict pass rate (Code fix)",1,"rate","100% (10/10)",10,[0.7225,1],"ci95","Claude Code · same prompt repeated 10 times","\u0001","\u0001","\u0001","\u0001"],["caching-consistency","consistency-distinct-answers","Exact number","Claude Sonnet 5.5 · Claude Code","consistency-distinct-answers/Exact number","Same prompt, 10 times: how many different answers (Exact number)",1,"count","1",10,"\u0001","\u0001","Claude Code · same prompt repeated 10 times","\u0001","\u0001","\u0001","\u0001"],["caching-consistency","consistency-distinct-answers","JSON object","Claude Sonnet 5.5 · Claude Code","consistency-distinct-answers/JSON object","Same prompt, 10 times: how many different answers (JSON object)",1,"count","1",10,"\u0001","\u0001","Claude Code · same prompt repeated 10 times","\u0001","\u0001","\u0001","\u0001"],["caching-consistency","consistency-distinct-answers","Code fix","Claude Sonnet 5.5 · Claude Code","consistency-distinct-answers/Code fix","Same prompt, 10 times: how many different answers (Code fix)",3,"count","3",10,"\u0001","\u0001","Claude Code · same prompt repeated 10 times","\u0001","\u0001","\u0001","\u0001"],["caching-consistency","consistency-latency-spread","Exact number","Claude Sonnet 5.5 · Claude Code","consistency-latency-spread/Exact number","Same prompt, 10 times: time per call (Exact number)",6.89,"seconds","6.89 s",10,"\u0001","minmax","Claude Code · same prompt repeated 10 times","\u0001",[5.81,7.81],"\u0001","\u0001"],["caching-consistency","consistency-latency-spread","JSON object","Claude Sonnet 5.5 · Claude Code","consistency-latency-spread/JSON object","Same prompt, 10 times: time per call (JSON object)",2.89,"seconds","2.89 s",10,"\u0001","minmax","Claude Code · same prompt repeated 10 times","\u0001",[2.68,5.3],"\u0001","\u0001"],["caching-consistency","consistency-latency-spread","Code fix","Claude Sonnet 5.5 · Claude Code","consistency-latency-spread/Code fix","Same prompt, 10 times: time per call (Code fix)",2.67,"seconds","2.67 s",10,"\u0001","minmax","Claude Code · same prompt repeated 10 times","\u0001",[2.32,4.34],"\u0001","\u0001"],["agent-memory","memory-full-pass","Claude Sonnet 5.5","No memory","memory-full-pass/No memory","Full pass rate by kind of memory: No memory",0.6,"rate","60% (9/15)",15,[0.3575,0.8018],"ci95","","\u0001","\u0001","\u0001","\u0001"],["agent-memory","memory-full-pass","Claude Sonnet 5.5","/init CLAUDE.md","memory-full-pass//init CLAUDE.md","Full pass rate by kind of memory: /init CLAUDE.md",0.6,"rate","60% (9/15)",15,[0.3575,0.8018],"ci95","","\u0001","\u0001","\u0001","\u0001"],["agent-memory","memory-full-pass","Claude Sonnet 5.5","Curated, 11 lines","memory-full-pass/Curated, 11 lines","Full pass rate by kind of memory: Curated, 11 lines",1,"rate","100% (15/15)",15,[0.7961,1],"ci95","","\u0001","\u0001","\u0001","\u0001"],["agent-memory","memory-full-pass","Claude Sonnet 5.5","Raw notes, 60 lines","memory-full-pass/Raw notes, 60 lines","Full pass rate by kind of memory: Raw notes, 60 lines",0.9333,"rate","93% (14/15)",15,[0.7018,0.9881],"ci95","","\u0001","\u0001","\u0001","\u0001"],["agent-memory","memory-full-pass","Claude Sonnet 5.5","Dreamed notes","memory-full-pass/Dreamed notes","Full pass rate by kind of memory: Dreamed notes",1,"rate","100% (15/15)",15,[0.7961,1],"ci95","","\u0001","\u0001","\u0001","\u0001"],["agent-memory","memory-full-pass","Claude Sonnet 5.5","Handbook, 210 lines","memory-full-pass/Handbook, 210 lines","Full pass rate by kind of memory: Handbook, 210 lines",1,"rate","100% (15/15)",15,[0.7961,1],"ci95","","\u0001","\u0001","\u0001","\u0001"],["agent-memory","memory-full-pass","Claude Sonnet 5.5","Stop hook only","memory-full-pass/Stop hook only","Full pass rate by kind of memory: Stop hook only",0.8,"rate","80% (12/15)",15,[0.5481,0.9295],"ci95","","\u0001","\u0001","\u0001","\u0001"],["agent-memory","memory-full-pass","Claude Sonnet 5.5","Curated + hook","memory-full-pass/Curated + hook","Full pass rate by kind of memory: Curated + hook",1,"rate","100% (15/15)",15,[0.7961,1],"ci95","","\u0001","\u0001","\u0001","\u0001"],["agent-memory","memory-team-knowledge-by-model","Claude Sonnet 5.5","No memory","memory-team-knowledge-by-model/No memory","Team knowledge followed, Sonnet vs Haiku: No memory",0.4,"rate","40% (6/15)",15,[0.1982,0.6425],"ci95","","\u0001","\u0001","\u0001","\u0001"],["agent-memory","memory-team-knowledge-by-model","Claude Sonnet 5.5","/init CLAUDE.md","memory-team-knowledge-by-model//init CLAUDE.md","Team knowledge followed, Sonnet vs Haiku: /init CLAUDE.md",0.6667,"rate","67% (10/15)",15,[0.4171,0.8482],"ci95","","\u0001","\u0001","\u0001","\u0001"],["agent-memory","memory-team-knowledge-by-model","Claude Sonnet 5.5","Curated, 11 lines","memory-team-knowledge-by-model/Curated, 11 lines","Team knowledge followed, Sonnet vs Haiku: Curated, 11 lines",1,"rate","100% (15/15)",15,[0.7961,1],"ci95","","\u0001","\u0001","\u0001","\u0001"],["agent-memory","memory-team-knowledge-by-model","Claude Sonnet 5.5","Raw notes, 60 lines","memory-team-knowledge-by-model/Raw notes, 60 lines","Team knowledge followed, Sonnet vs Haiku: Raw notes, 60 lines",1,"rate","100% (15/15)",15,[0.7961,1],"ci95","","\u0001","\u0001","\u0001","\u0001"],["agent-memory","memory-team-knowledge-by-model","Claude Sonnet 5.5","Dreamed notes","memory-team-knowledge-by-model/Dreamed notes","Team knowledge followed, Sonnet vs Haiku: Dreamed notes",1,"rate","100% (15/15)",15,[0.7961,1],"ci95","","\u0001","\u0001","\u0001","\u0001"],["agent-memory","memory-team-knowledge-by-model","Claude Sonnet 5.5","Handbook, 210 lines","memory-team-knowledge-by-model/Handbook, 210 lines","Team knowledge followed, Sonnet vs Haiku: Handbook, 210 lines",1,"rate","100% (15/15)",15,[0.7961,1],"ci95","","\u0001","\u0001","\u0001","\u0001"],["agent-memory","memory-team-knowledge-by-model","Claude Sonnet 5.5","Stop hook only","memory-team-knowledge-by-model/Stop hook only","Team knowledge followed, Sonnet vs Haiku: Stop hook only",0.6667,"rate","67% (10/15)",15,[0.4171,0.8482],"ci95","","\u0001","\u0001","\u0001","\u0001"],["agent-memory","memory-team-knowledge-by-model","Claude Sonnet 5.5","Curated + hook","memory-team-knowledge-by-model/Curated + hook","Team knowledge followed, Sonnet vs Haiku: Curated + hook",1,"rate","100% (15/15)",15,[0.7961,1],"ci95","","\u0001","\u0001","\u0001","\u0001"],["agent-memory","memory-broken-test-command","Claude Sonnet 5.5","No memory","memory-broken-test-command/No memory","A stale README command: who still ran it?: No memory",0.8,"rate","80% (12/15)",15,[0.5481,0.9295],"ci95","","\u0001","\u0001","lower","\u0001"],["agent-memory","memory-broken-test-command","Claude Sonnet 5.5","/init CLAUDE.md","memory-broken-test-command//init CLAUDE.md","A stale README command: who still ran it?: /init CLAUDE.md",0.8667,"rate","87% (13/15)",15,[0.6212,0.9626],"ci95","","\u0001","\u0001","lower","\u0001"],["agent-memory","memory-broken-test-command","Claude Sonnet 5.5","Curated, 11 lines","memory-broken-test-command/Curated, 11 lines","A stale README command: who still ran it?: Curated, 11 lines",0,"rate","0% (0/15)",15,[0,0.2039],"ci95","","\u0001","\u0001","lower","\u0001"],["agent-memory","memory-broken-test-command","Claude Sonnet 5.5","Raw notes, 60 lines","memory-broken-test-command/Raw notes, 60 lines","A stale README command: who still ran it?: Raw notes, 60 lines",0,"rate","0% (0/15)",15,[0,0.2039],"ci95","","\u0001","\u0001","lower","\u0001"],["agent-memory","memory-broken-test-command","Claude Sonnet 5.5","Dreamed notes","memory-broken-test-command/Dreamed notes","A stale README command: who still ran it?: Dreamed notes",0,"rate","0% (0/15)",15,[0,0.2039],"ci95","","\u0001","\u0001","lower","\u0001"],["agent-memory","memory-broken-test-command","Claude Sonnet 5.5","Handbook, 210 lines","memory-broken-test-command/Handbook, 210 lines","A stale README command: who still ran it?: Handbook, 210 lines",0,"rate","0% (0/15)",15,[0,0.2039],"ci95","","\u0001","\u0001","lower","\u0001"],["agent-memory","memory-broken-test-command","Claude Sonnet 5.5","Stop hook only","memory-broken-test-command/Stop hook only","A stale README command: who still ran it?: Stop hook only",0.6,"rate","60% (9/15)",15,[0.3575,0.8018],"ci95","","\u0001","\u0001","lower","\u0001"],["agent-memory","memory-broken-test-command","Claude Sonnet 5.5","Curated + hook","memory-broken-test-command/Curated + hook","A stale README command: who still ran it?: Curated + hook",0,"rate","0% (0/15)",15,[0,0.2039],"ci95","","\u0001","\u0001","lower","\u0001"],["agent-memory","memory-cost-per-full-pass","Claude Sonnet 5.5","No memory","memory-cost-per-full-pass/No memory","List-price cost per fully correct result (calculation): No memory",0.1386,"usd","$0.14",9,"\u0001","\u0001","",true,"\u0001","\u0001","\u0001"],["agent-memory","memory-cost-per-full-pass","Claude Sonnet 5.5","/init CLAUDE.md","memory-cost-per-full-pass//init CLAUDE.md","List-price cost per fully correct result (calculation): /init CLAUDE.md",0.1359,"usd","$0.14",9,"\u0001","\u0001","",true,"\u0001","\u0001","\u0001"],["agent-memory","memory-cost-per-full-pass","Claude Sonnet 5.5","Curated, 11 lines","memory-cost-per-full-pass/Curated, 11 lines","List-price cost per fully correct result (calculation): Curated, 11 lines",0.0818,"usd","$0.082",15,"\u0001","\u0001","",true,"\u0001","\u0001","\u0001"],["agent-memory","memory-cost-per-full-pass","Claude Sonnet 5.5","Raw notes, 60 lines","memory-cost-per-full-pass/Raw notes, 60 lines","List-price cost per fully correct result (calculation): Raw notes, 60 lines",0.1009,"usd","$0.10",14,"\u0001","\u0001","",true,"\u0001","\u0001","\u0001"],["agent-memory","memory-cost-per-full-pass","Claude Sonnet 5.5","Dreamed notes","memory-cost-per-full-pass/Dreamed notes","List-price cost per fully correct result (calculation): Dreamed notes",0.0896,"usd","$0.090",15,"\u0001","\u0001","",true,"\u0001","\u0001","\u0001"],["agent-memory","memory-cost-per-full-pass","Claude Sonnet 5.5","Handbook, 210 lines","memory-cost-per-full-pass/Handbook, 210 lines","List-price cost per fully correct result (calculation): Handbook, 210 lines",0.1006,"usd","$0.10",15,"\u0001","\u0001","",true,"\u0001","\u0001","\u0001"],["agent-memory","memory-cost-per-full-pass","Claude Sonnet 5.5","Stop hook only","memory-cost-per-full-pass/Stop hook only","List-price cost per fully correct result (calculation): Stop hook only",0.1278,"usd","$0.13",12,"\u0001","\u0001","",true,"\u0001","\u0001","\u0001"],["agent-memory","memory-cost-per-full-pass","Claude Sonnet 5.5","Curated + hook","memory-cost-per-full-pass/Curated + hook","List-price cost per fully correct result (calculation): Curated + hook",0.0843,"usd","$0.084",15,"\u0001","\u0001","",true,"\u0001","\u0001","\u0001"],["agent-memory","memory-wall-time","Claude Sonnet 5.5","No memory","memory-wall-time/No memory","Time per session: No memory",18,"seconds","18.0 s",15,"\u0001","\u0001","","\u0001","\u0001","\u0001","\u0001"],["agent-memory","memory-wall-time","Claude Sonnet 5.5","/init CLAUDE.md","memory-wall-time//init CLAUDE.md","Time per session: /init CLAUDE.md",19,"seconds","19.0 s",15,"\u0001","\u0001","","\u0001","\u0001","\u0001","\u0001"],["agent-memory","memory-wall-time","Claude Sonnet 5.5","Curated, 11 lines","memory-wall-time/Curated, 11 lines","Time per session: Curated, 11 lines",21.9,"seconds","21.9 s",15,"\u0001","\u0001","","\u0001","\u0001","\u0001","\u0001"],["agent-memory","memory-wall-time","Claude Sonnet 5.5","Raw notes, 60 lines","memory-wall-time/Raw notes, 60 lines","Time per session: Raw notes, 60 lines",26.8,"seconds","26.8 s",15,"\u0001","\u0001","","\u0001","\u0001","\u0001","\u0001"],["agent-memory","memory-wall-time","Claude Sonnet 5.5","Dreamed notes","memory-wall-time/Dreamed notes","Time per session: Dreamed notes",27.2,"seconds","27.2 s",15,"\u0001","\u0001","","\u0001","\u0001","\u0001","\u0001"],["agent-memory","memory-wall-time","Claude Sonnet 5.5","Handbook, 210 lines","memory-wall-time/Handbook, 210 lines","Time per session: Handbook, 210 lines",23.6,"seconds","23.6 s",15,"\u0001","\u0001","","\u0001","\u0001","\u0001","\u0001"],["agent-memory","memory-wall-time","Claude Sonnet 5.5","Stop hook only","memory-wall-time/Stop hook only","Time per session: Stop hook only",27.3,"seconds","27.3 s",15,"\u0001","\u0001","","\u0001","\u0001","\u0001","\u0001"],["agent-memory","memory-wall-time","Claude Sonnet 5.5","Curated + hook","memory-wall-time/Curated + hook","Time per session: Curated + hook",22,"seconds","22.0 s",15,"\u0001","\u0001","","\u0001","\u0001","\u0001","\u0001"],["routing-jev-vs-llm","routing-exact-decisions","Exact rate","Claude Sonnet 5.5","routing-exact-decisions","Typed routing decisions answered exactly right",0.939,"rate","94% (77/82)",82,[0.8651,0.9737],"ci95","typed routing decisions · via Claude Code","\u0001","\u0001","\u0001","\u0001"],["routing-jev-vs-llm","routing-key-accuracy","Key accuracy","Claude Sonnet 5.5","routing-key-accuracy","Per-question accuracy",0.9742,"rate","97% (189/194)",194,[0.9411,0.9889],"ci95","typed routing decisions · via Claude Code","\u0001","\u0001","\u0001","\u0001"],["routing-jev-vs-llm","routing-exact-by-decision","Claude Sonnet 5.5","Failure class","routing-exact-by-decision/Failure class","Exact rate by decision type: Failure class",1,"rate","100% (18/18)",18,[0.8241,1],"ci95","typed routing decisions · via Claude Code","\u0001","\u0001","\u0001","\u0001"],["routing-jev-vs-llm","routing-exact-by-decision","Claude Sonnet 5.5","Message intent","routing-exact-by-decision/Message intent","Exact rate by decision type: Message intent",1,"rate","100% (20/20)",20,[0.8389,1],"ci95","typed routing decisions · via Claude Code","\u0001","\u0001","\u0001","\u0001"],["routing-jev-vs-llm","routing-exact-by-decision","Claude Sonnet 5.5","Is it a rule?","routing-exact-by-decision/Is it a rule?","Exact rate by decision type: Is it a rule?",1,"rate","100% (12/12)",12,[0.7575,1],"ci95","typed routing decisions · via Claude Code","\u0001","\u0001","\u0001","\u0001"],["routing-jev-vs-llm","routing-exact-by-decision","Claude Sonnet 5.5","Context shape","routing-exact-by-decision/Context shape","Exact rate by decision type: Context shape",0.8438,"rate","84% (27/32)",32,[0.6825,0.9314],"ci95","typed routing decisions · via Claude Code","\u0001","\u0001","\u0001","\u0001"],["routing-jev-vs-llm","routing-cost-per-1000","Cost","Claude Sonnet 5.5","routing-cost-per-1000","Cost per 1,000 routing decisions",4.996,"usd","$5.00",82,"\u0001","\u0001","typed routing decisions · via Claude Code",true,"\u0001","\u0001","\u0001"],["routing-jev-vs-llm","routing-decision-latency","Wall time (CLI)","Claude Sonnet 5.5","routing-decision-latency/Wall time (CLI)","Time per routing decision (Wall time (CLI))",2598,"ms","2,598 ms",82,"\u0001","p50-p95","typed routing decisions · via Claude Code","\u0001",[2598,4298],"\u0001","\u0001"],["routing-jev-vs-llm","routing-decision-latency","Model time (API)","Claude Sonnet 5.5","routing-decision-latency/Model time (API)","Time per routing decision (Model time (API))",1599,"ms","1,599 ms",82,"\u0001","p50-p95","typed routing decisions · via Claude Code","\u0001",[1599,2574],"\u0001","\u0001"],["routing-overhead","router-overhead-decision-latency","Decision time","Claude Sonnet 5.5 (effort low, via Claude Code)","router-overhead-decision-latency","Time to make one routing decision",2597,"ms","2,597 ms",82,"\u0001","p50-p95","effort low · via Claude Code · routing overhead per decision","\u0001",[2597,4298],"\u0001","\u0001"],["routing-overhead","router-overhead-cli-vs-model-time","Model API time","Claude Sonnet 5.5 (effort low, via Claude Code)","router-overhead-cli-vs-model-time/Model API time","Where an LLM router’s time goes: model vs CLI (Model API time)",1596,"ms","1,596 ms",82,"\u0001","p50-p95","effort low · via Claude Code · routing overhead per decision","\u0001",[1596,2583],"\u0001","\u0001"],["routing-overhead","router-overhead-cli-vs-model-time","CLI and harness time","Claude Sonnet 5.5 (effort low, via Claude Code)","router-overhead-cli-vs-model-time/CLI and harness time","Where an LLM router’s time goes: model vs CLI (CLI and harness time)",973,"ms","973 ms",82,"\u0001","p50-p95","effort low · via Claude Code · routing overhead per decision","\u0001",[973,1277],"\u0001","\u0001"],["routing-overhead","router-overhead-completed","Completed","Claude Sonnet 5.5 (effort low, via Claude Code)","router-overhead-completed","Routing calls that returned a decision",1,"rate","100% (82/82)",82,[0.9552,1],"ci95","effort low · via Claude Code · routing overhead per decision","\u0001","\u0001","\u0001","\u0001"],["routing-overhead","router-overhead-cost-list-price","Cost per 1,000 decisions (list price)","Claude Sonnet 5.5 (effort low, via Claude Code)","router-overhead-cost-list-price","Cost per 1,000 routing decisions for the model routers (calculation)",4.996,"usd","$5.00",82,"\u0001","\u0001","effort low · via Claude Code · calculation: Agent’s recorded tokens at this model’s list price · routing overhead per decision",true,"\u0001","\u0001","\u0001"],["routing-overhead","router-overhead-cost-per-1000-tasks","Every model call routed (49.5 per task)","Claude Sonnet 5.5 (effort low, via Claude Code)","router-overhead-cost-per-1000-tasks/Every model call routed (49.5 per task)","Added routing cost per 1,000 tasks (calculation) (Every model call routed (49.5 per task))",247.3,"usd","$247.30","\u0001","\u0001","\u0001","effort low · via Claude Code · calculation per 1,000 tasks from recorded decision counts",true,"\u0001","\u0001","\u0001"],["routing-overhead","router-overhead-cost-per-1000-tasks","Only System One decisions (7 per task)","Claude Sonnet 5.5 (effort low, via Claude Code)","router-overhead-cost-per-1000-tasks/Only System One decisions (7 per task)","Added routing cost per 1,000 tasks (calculation) (Only System One decisions (7 per task))",34.97,"usd","$34.97","\u0001","\u0001","\u0001","effort low · via Claude Code · calculation per 1,000 tasks from recorded decision counts",true,"\u0001","\u0001","\u0001"],["routing-overhead","router-overhead-delay-per-task","Every model call routed (49.5 per task)","Claude Sonnet 5.5 (effort low, via Claude Code)","router-overhead-delay-per-task/Every model call routed (49.5 per task)","Added routing delay per task (calculation) (Every model call routed (49.5 per task))",128.5515,"seconds","128.6 s","\u0001","\u0001","\u0001","effort low · via Claude Code · calculation per task from recorded decision counts, decisions in line",true,"\u0001","\u0001","\u0001"],["routing-overhead","router-overhead-delay-per-task","Only System One decisions (7 per task)","Claude Sonnet 5.5 (effort low, via Claude Code)","router-overhead-delay-per-task/Only System One decisions (7 per task)","Added routing delay per task (calculation) (Only System One decisions (7 per task))",18.179,"seconds","18.2 s","\u0001","\u0001","\u0001","effort low · via Claude Code · calculation per task from recorded decision counts, decisions in line",true,"\u0001","\u0001","\u0001"],["cost-thought-experiments","repriced-cost-per-resolved","Repriced cost per resolved instance","Claude Sonnet 5.5","repriced-cost-per-resolved","Thought experiment: the same tokens at other list prices",3.489,"usd","$3.49","\u0001","\u0001","\u0001","calculation: Agent’s recorded tokens at this model’s list price",true,"\u0001","\u0001","\u0001"],["cost-thought-experiments","prompt-cache-savings","With caching (as recorded)","Claude Sonnet 5.5","prompt-cache-savings/With caching (as recorded)","Thought experiment: what prompt caching saved (With caching (as recorded))",87.23,"usd","$87.23","\u0001","\u0001","\u0001","calculation: Agent’s recorded tokens at this model’s list price",true,"\u0001","\u0001","\u0001"],["cost-thought-experiments","prompt-cache-savings","Without caching","Claude Sonnet 5.5","prompt-cache-savings/Without caching","Thought experiment: what prompt caching saved (Without caching)",343.33,"usd","$343.33","\u0001","\u0001","\u0001","calculation: Agent’s recorded tokens at this model’s list price",true,"\u0001","\u0001","\u0001"],["cli-model-latency-tokens","scheduler-repair-claude-vs-codex","Total time","Claude Code CLI · Sonnet 5.5 · medium","scheduler-repair-claude-vs-codex/Total time","Repairing a scheduler: Claude Code vs Codex vs API (Total time)",15,"seconds","15.0 s",3,"\u0001","minmax","Claude Code CLI · effort medium · scheduler repair, 296 checks, 3 runs","\u0001",[13.89,15.89],"\u0001","\u0001"],["cli-model-latency-tokens","scheduler-repair-claude-vs-codex","First useful output","Claude Code CLI · Sonnet 5.5 · medium","scheduler-repair-claude-vs-codex/First useful output","Repairing a scheduler: Claude Code vs Codex vs API (First useful output)",7.55,"seconds","7.55 s",3,"\u0001","minmax","Claude Code CLI · effort medium · scheduler repair, 296 checks, 3 runs","\u0001",[6.77,7.63],"\u0001","\u0001"],["cli-model-latency-tokens","scheduler-repair-output-tokens","Output tokens","Claude Code CLI · Sonnet 5.5 · medium","scheduler-repair-output-tokens/Output tokens","Output tokens to repair the scheduler (Output tokens)",2227,"tokens","2,227",3,"\u0001","\u0001","Claude Code CLI · effort medium · scheduler repair, 296 checks, 3 runs","\u0001","\u0001","\u0001","\u0001"],["single-call-vs-agent-loop","agent-loop-pass-rate","Strict pass","Claude Sonnet 5.5 (single call) · Claude Code","agent-loop-pass-rate","Strict pass rate: single call vs agent loop on eight hard tasks",1,"rate","100% (24/24)",24,[0.862,1],"ci95","Claude Code · single call","\u0001","\u0001","higher","\u0001"],["single-call-vs-agent-loop","agent-loop-pass-rate","Strict pass","Claude Sonnet 5.5 (agent loop) · Claude Code","agent-loop-pass-rate","Strict pass rate: single call vs agent loop on eight hard tasks",1,"rate","100% (16/16)",16,[0.8064,1],"ci95","Claude Code · agent loop","\u0001","\u0001","higher","\u0001"],["single-call-vs-agent-loop","agent-loop-by-task","Claude Sonnet 5.5 (single call) · Claude Code","Interval merge fix","agent-loop-by-task/Interval merge fix","Strict passes per task: single call vs agent loop: Interval merge fix",1,"rate","100% (3/3)",3,[0.4385,1],"ci95","Claude Code · single call","\u0001","\u0001","higher","\u0001"],["single-call-vs-agent-loop","agent-loop-by-task","Claude Sonnet 5.5 (single call) · Claude Code","DST day-length fix","agent-loop-by-task/DST day-length fix","Strict passes per task: single call vs agent loop: DST day-length fix",1,"rate","100% (3/3)",3,[0.4385,1],"ci95","Claude Code · single call","\u0001","\u0001","higher","\u0001"],["single-call-vs-agent-loop","agent-loop-by-task","Claude Sonnet 5.5 (single call) · Claude Code","CSV parser","agent-loop-by-task/CSV parser","Strict passes per task: single call vs agent loop: CSV parser",1,"rate","100% (3/3)",3,[0.4385,1],"ci95","Claude Code · single call","\u0001","\u0001","higher","\u0001"],["single-call-vs-agent-loop","agent-loop-by-task","Claude Sonnet 5.5 (single call) · Claude Code","Event-loop order","agent-loop-by-task/Event-loop order","Strict passes per task: single call vs agent loop: Event-loop order",1,"rate","100% (3/3)",3,[0.4385,1],"ci95","Claude Code · single call","\u0001","\u0001","higher","\u0001"],["single-call-vs-agent-loop","agent-loop-by-task","Claude Sonnet 5.5 (single call) · Claude Code","Room schedule","agent-loop-by-task/Room schedule","Strict passes per task: single call vs agent loop: Room schedule",1,"rate","100% (3/3)",3,[0.4385,1],"ci95","Claude Code · single call","\u0001","\u0001","higher","\u0001"],["single-call-vs-agent-loop","agent-loop-by-task","Claude Sonnet 5.5 (single call) · Claude Code","SemVer regex","agent-loop-by-task/SemVer regex","Strict passes per task: single call vs agent loop: SemVer regex",1,"rate","100% (3/3)",3,[0.4385,1],"ci95","Claude Code · single call","\u0001","\u0001","higher","\u0001"],["single-call-vs-agent-loop","agent-loop-by-task","Claude Sonnet 5.5 (single call) · Claude Code","Money refactor","agent-loop-by-task/Money refactor","Strict passes per task: single call vs agent loop: Money refactor",1,"rate","100% (3/3)",3,[0.4385,1],"ci95","Claude Code · single call","\u0001","\u0001","higher","\u0001"],["single-call-vs-agent-loop","agent-loop-by-task","Claude Sonnet 5.5 (single call) · Claude Code","SQL report","agent-loop-by-task/SQL report","Strict passes per task: single call vs agent loop: SQL report",1,"rate","100% (3/3)",3,[0.4385,1],"ci95","Claude Code · single call","\u0001","\u0001","higher","\u0001"],["single-call-vs-agent-loop","agent-loop-by-task","Claude Sonnet 5.5 (agent loop) · Claude Code","Interval merge fix","agent-loop-by-task/Interval merge fix","Strict passes per task: single call vs agent loop: Interval merge fix",1,"rate","100% (2/2)",2,[0.3424,1],"ci95","Claude Code · agent loop","\u0001","\u0001","higher","\u0001"],["single-call-vs-agent-loop","agent-loop-by-task","Claude Sonnet 5.5 (agent loop) · Claude Code","DST day-length fix","agent-loop-by-task/DST day-length fix","Strict passes per task: single call vs agent loop: DST day-length fix",1,"rate","100% (2/2)",2,[0.3424,1],"ci95","Claude Code · agent loop","\u0001","\u0001","higher","\u0001"],["single-call-vs-agent-loop","agent-loop-by-task","Claude Sonnet 5.5 (agent loop) · Claude Code","CSV parser","agent-loop-by-task/CSV parser","Strict passes per task: single call vs agent loop: CSV parser",1,"rate","100% (2/2)",2,[0.3424,1],"ci95","Claude Code · agent loop","\u0001","\u0001","higher","\u0001"],["single-call-vs-agent-loop","agent-loop-by-task","Claude Sonnet 5.5 (agent loop) · Claude Code","Event-loop order","agent-loop-by-task/Event-loop order","Strict passes per task: single call vs agent loop: Event-loop order",1,"rate","100% (2/2)",2,[0.3424,1],"ci95","Claude Code · agent loop","\u0001","\u0001","higher","\u0001"],["single-call-vs-agent-loop","agent-loop-by-task","Claude Sonnet 5.5 (agent loop) · Claude Code","Room schedule","agent-loop-by-task/Room schedule","Strict passes per task: single call vs agent loop: Room schedule",1,"rate","100% (2/2)",2,[0.3424,1],"ci95","Claude Code · agent loop","\u0001","\u0001","higher","\u0001"],["single-call-vs-agent-loop","agent-loop-by-task","Claude Sonnet 5.5 (agent loop) · Claude Code","SemVer regex","agent-loop-by-task/SemVer regex","Strict passes per task: single call vs agent loop: SemVer regex",1,"rate","100% (2/2)",2,[0.3424,1],"ci95","Claude Code · agent loop","\u0001","\u0001","higher","\u0001"],["single-call-vs-agent-loop","agent-loop-by-task","Claude Sonnet 5.5 (agent loop) · Claude Code","Money refactor","agent-loop-by-task/Money refactor","Strict passes per task: single call vs agent loop: Money refactor",1,"rate","100% (2/2)",2,[0.3424,1],"ci95","Claude Code · agent loop","\u0001","\u0001","higher","\u0001"],["single-call-vs-agent-loop","agent-loop-by-task","Claude Sonnet 5.5 (agent loop) · Claude Code","SQL report","agent-loop-by-task/SQL report","Strict passes per task: single call vs agent loop: SQL report",1,"rate","100% (2/2)",2,[0.3424,1],"ci95","Claude Code · agent loop","\u0001","\u0001","higher","\u0001"],["single-call-vs-agent-loop","agent-loop-total-time","Total time per attempt","Claude Sonnet 5.5 (single call) · Claude Code","agent-loop-total-time","Total time per attempt: single call vs agent loop",7.75,"seconds","7.75 s",24,"\u0001","minmax","Claude Code · single call","\u0001",[2.26,34.79],"\u0001","\u0001"],["single-call-vs-agent-loop","agent-loop-total-time","Total time per attempt","Claude Sonnet 5.5 (agent loop) · Claude Code","agent-loop-total-time","Total time per attempt: single call vs agent loop",7.41,"seconds","7.41 s",16,"\u0001","minmax","Claude Code · agent loop","\u0001",[2.75,24.19],"\u0001","\u0001"],["single-call-vs-agent-loop","agent-loop-tokens","Input tokens (cache reads included)","Claude Sonnet 5.5 (single call) · Claude Code","agent-loop-tokens/Input tokens (cache reads included)","Tokens per attempt: single call vs agent loop (Input tokens (cache reads included))",2281,"tokens","2,281",24,"\u0001","minmax","Claude Code · single call","\u0001",[2234,2669],"\u0001","\u0001"],["single-call-vs-agent-loop","agent-loop-tokens","Input tokens (cache reads included)","Claude Sonnet 5.5 (agent loop) · Claude Code","agent-loop-tokens/Input tokens (cache reads included)","Tokens per attempt: single call vs agent loop (Input tokens (cache reads included))",9550,"tokens","9,550",16,"\u0001","minmax","Claude Code · agent loop","\u0001",[9398,33040],"\u0001","\u0001"],["single-call-vs-agent-loop","agent-loop-tokens","Output tokens","Claude Sonnet 5.5 (single call) · Claude Code","agent-loop-tokens/Output tokens","Tokens per attempt: single call vs agent loop (Output tokens)",1050,"tokens","1,050",24,"\u0001","minmax","Claude Code · single call","\u0001",[176,3895],"\u0001","\u0001"],["single-call-vs-agent-loop","agent-loop-tokens","Output tokens","Claude Sonnet 5.5 (agent loop) · Claude Code","agent-loop-tokens/Output tokens","Tokens per attempt: single call vs agent loop (Output tokens)",876,"tokens","876",16,"\u0001","minmax","Claude Code · agent loop","\u0001",[219,3243],"\u0001","\u0001"],["single-call-vs-agent-loop","agent-loop-tool-calls","Tool calls per attempt","Claude Sonnet 5.5 (agent loop) · Claude Code","agent-loop-tool-calls","Tool calls per agent-loop attempt",0,"count","0",16,"\u0001","minmax","Claude Code · agent loop","\u0001",[0,3],"\u0001","\u0001"],["single-call-vs-agent-loop","agent-loop-cost-per-pass","Cost per strict pass","Claude Sonnet 5.5 (single call) · Claude Code","agent-loop-cost-per-pass","List-price cost per strict pass: single call vs agent loop (calculation)",0.01435,"usd","$0.014",24,"\u0001","\u0001","Claude Code · single call",true,"\u0001","\u0001","\u0001"],["single-call-vs-agent-loop","agent-loop-cost-per-pass","Cost per strict pass","Claude Sonnet 5.5 (agent loop) · Claude Code","agent-loop-cost-per-pass","List-price cost per strict pass: single call vs agent loop (calculation)",0.02746,"usd","$0.027",16,"\u0001","\u0001","Claude Code · agent loop",true,"\u0001","\u0001","\u0001"],["haiku-thinking-on-off","haiku-thinking-router-exact","Exact decisions (every scored question right)","Claude Sonnet 5.5 (low) · Claude Code","haiku-thinking-router-exact/Exact decisions (every scored question right)","Haiku thinking study: typed routing decisions answered exactly right (Exact decisions (every scored question right))",0.939,"rate","94% (77/82)",82,[0.8651,0.9737],"ci95","Claude Code · effort low · typed routing decisions, thinking on vs off","\u0001","\u0001","higher","\u0001"],["haiku-thinking-on-off","haiku-thinking-router-exact","Per-question accuracy","Claude Sonnet 5.5 (low) · Claude Code","haiku-thinking-router-exact/Per-question accuracy","Haiku thinking study: typed routing decisions answered exactly right (Per-question accuracy)",0.9742,"rate","97% (189/194)",194,[0.9411,0.9889],"ci95","Claude Code · effort low · typed routing decisions, thinking on vs off","\u0001","\u0001","higher","\u0001"],["haiku-thinking-on-off","haiku-thinking-router-latency","Wall time (CLI)","Claude Sonnet 5.5 (low) · Claude Code","haiku-thinking-router-latency/Wall time (CLI)","Haiku thinking study: time per routing decision (Wall time (CLI))",2.6,"seconds","2.60 s",82,"\u0001","p50-p95","Claude Code · effort low · typed routing decisions, thinking on vs off","\u0001",[2.6,4.3],"\u0001","\u0001"],["haiku-thinking-on-off","haiku-thinking-router-latency","Model time (API)","Claude Sonnet 5.5 (low) · Claude Code","haiku-thinking-router-latency/Model time (API)","Haiku thinking study: time per routing decision (Model time (API))",1.6,"seconds","1.60 s",82,"\u0001","p50-p95","Claude Code · effort low · typed routing decisions, thinking on vs off","\u0001",[1.6,2.58],"\u0001","\u0001"],["haiku-thinking-on-off","haiku-thinking-router-tokens","Thinking tokens","Claude Sonnet 5.5 (low) · Claude Code","haiku-thinking-router-tokens/Thinking tokens","Haiku thinking study: thinking and visible output tokens per routing decision (Thinking tokens)",2,"tokens","2",82,"\u0001","\u0001","Claude Code · effort low · typed routing decisions, thinking on vs off","\u0001","\u0001","\u0001","\u0001"],["haiku-thinking-on-off","haiku-thinking-router-tokens","Visible output tokens","Claude Sonnet 5.5 (low) · Claude Code","haiku-thinking-router-tokens/Visible output tokens","Haiku thinking study: thinking and visible output tokens per routing decision (Visible output tokens)",105,"tokens","105",82,"\u0001","\u0001","Claude Code · effort low · typed routing decisions, thinking on vs off","\u0001","\u0001","\u0001","\u0001"],["haiku-thinking-on-off","haiku-thinking-router-cost","Cost per 1,000 decisions","Claude Sonnet 5.5 (low) · Claude Code","haiku-thinking-router-cost","Haiku thinking study: list-price cost per 1,000 routing decisions (calculation)",7.324,"usd","$7.32",82,"\u0001","\u0001","Claude Code · effort low · typed routing decisions, thinking on vs off",true,"\u0001","\u0001","\u0001"],["json-schema-vs-instructions","structured-output-pass-rate","Strict pass: the whole reply is the right JSON","Claude Sonnet 5.5 (instructions) · Claude Code","structured-output-pass-rate/Strict pass: the whole reply is the right JSON","Does a JSON schema raise the pass rate? Instructions vs schema mode (Strict pass: the whole reply is the right JSON)",1,"rate","100% (12/12)",12,[0.7575,1],"ci95","Claude Code · instructions",true,"\u0001","higher","\u0001"],["json-schema-vs-instructions","structured-output-pass-rate","Strict pass: the whole reply is the right JSON","Claude Sonnet 5.5 (JSON schema) · Claude Code","structured-output-pass-rate/Strict pass: the whole reply is the right JSON","Does a JSON schema raise the pass rate? Instructions vs schema mode (Strict pass: the whole reply is the right JSON)",1,"rate","100% (12/12)",12,[0.7575,1],"ci95","Claude Code · JSON schema",true,"\u0001","higher","\u0001"],["json-schema-vs-instructions","structured-output-pass-rate","Right answer in any format (strict pass or format miss)","Claude Sonnet 5.5 (instructions) · Claude Code","structured-output-pass-rate/Right answer in any format (strict pass or format miss)","Does a JSON schema raise the pass rate? Instructions vs schema mode (Right answer in any format (strict pass or format miss))",1,"rate","100% (12/12)",12,[0.7575,1],"ci95","Claude Code · instructions",true,"\u0001","higher","\u0001"],["json-schema-vs-instructions","structured-output-pass-rate","Right answer in any format (strict pass or format miss)","Claude Sonnet 5.5 (JSON schema) · Claude Code","structured-output-pass-rate/Right answer in any format (strict pass or format miss)","Does a JSON schema raise the pass rate? Instructions vs schema mode (Right answer in any format (strict pass or format miss))",1,"rate","100% (12/12)",12,[0.7575,1],"ci95","Claude Code · JSON schema",true,"\u0001","higher","\u0001"],["json-schema-vs-instructions","structured-output-outcomes","Strict pass","Claude Sonnet 5.5 (instructions) · Claude Code","structured-output-outcomes/Strict pass","What each call produced: strict pass, format miss, wrong values or error (Strict pass)",12,"count","12",12,"\u0001","\u0001","Claude Code · instructions","\u0001","\u0001","\u0001","\u0001"],["json-schema-vs-instructions","structured-output-outcomes","Strict pass","Claude Sonnet 5.5 (JSON schema) · Claude Code","structured-output-outcomes/Strict pass","What each call produced: strict pass, format miss, wrong values or error (Strict pass)",12,"count","12",12,"\u0001","\u0001","Claude Code · JSON schema","\u0001","\u0001","\u0001","\u0001"],["json-schema-vs-instructions","structured-output-outcomes","Format miss","Claude Sonnet 5.5 (instructions) · Claude Code","structured-output-outcomes/Format miss","What each call produced: strict pass, format miss, wrong values or error (Format miss)",0,"count","0",12,"\u0001","\u0001","Claude Code · instructions","\u0001","\u0001","\u0001","\u0001"],["json-schema-vs-instructions","structured-output-outcomes","Format miss","Claude Sonnet 5.5 (JSON schema) · Claude Code","structured-output-outcomes/Format miss","What each call produced: strict pass, format miss, wrong values or error (Format miss)",0,"count","0",12,"\u0001","\u0001","Claude Code · JSON schema","\u0001","\u0001","\u0001","\u0001"],["json-schema-vs-instructions","structured-output-outcomes","Wrong values","Claude Sonnet 5.5 (instructions) · Claude Code","structured-output-outcomes/Wrong values","What each call produced: strict pass, format miss, wrong values or error (Wrong values)",0,"count","0",12,"\u0001","\u0001","Claude Code · instructions","\u0001","\u0001","\u0001","\u0001"],["json-schema-vs-instructions","structured-output-outcomes","Wrong values","Claude Sonnet 5.5 (JSON schema) · Claude Code","structured-output-outcomes/Wrong values","What each call produced: strict pass, format miss, wrong values or error (Wrong values)",0,"count","0",12,"\u0001","\u0001","Claude Code · JSON schema","\u0001","\u0001","\u0001","\u0001"],["json-schema-vs-instructions","structured-output-outcomes","Error","Claude Sonnet 5.5 (instructions) · Claude Code","structured-output-outcomes/Error","What each call produced: strict pass, format miss, wrong values or error (Error)",0,"count","0",12,"\u0001","\u0001","Claude Code · instructions","\u0001","\u0001","\u0001","\u0001"],["json-schema-vs-instructions","structured-output-outcomes","Error","Claude Sonnet 5.5 (JSON schema) · Claude Code","structured-output-outcomes/Error","What each call produced: strict pass, format miss, wrong values or error (Error)",0,"count","0",12,"\u0001","\u0001","Claude Code · JSON schema","\u0001","\u0001","\u0001","\u0001"],["json-schema-vs-instructions","structured-output-time","Median time per call (the three prompts pooled)","Claude Sonnet 5.5 (instructions) · Claude Code","structured-output-time","Time per call, instructions vs schema mode",3.52,"seconds","3.52 s",12,"\u0001","minmax","Claude Code · instructions","\u0001",[2.67,4.12],"\u0001","\u0001"],["json-schema-vs-instructions","structured-output-time","Median time per call (the three prompts pooled)","Claude Sonnet 5.5 (JSON schema) · Claude Code","structured-output-time","Time per call, instructions vs schema mode",4.2,"seconds","4.20 s",12,"\u0001","minmax","Claude Code · JSON schema","\u0001",[2.95,6.14],"\u0001","\u0001"],["json-schema-vs-instructions","structured-output-tokens","Median output tokens per call","Claude Sonnet 5.5 (instructions) · Claude Code","structured-output-tokens/Median output tokens per call","Output and reasoning tokens per call, instructions vs schema mode (Median output tokens per call)",368,"tokens","368",12,"\u0001","\u0001","Claude Code · instructions","\u0001","\u0001","\u0001","\u0001"],["json-schema-vs-instructions","structured-output-tokens","Median output tokens per call","Claude Sonnet 5.5 (JSON schema) · Claude Code","structured-output-tokens/Median output tokens per call","Output and reasoning tokens per call, instructions vs schema mode (Median output tokens per call)",424,"tokens","424",12,"\u0001","\u0001","Claude Code · JSON schema","\u0001","\u0001","\u0001","\u0001"],["prompt-cache-break-even","cache-break-even-reads","1-hour write (2× input), whole prefix new","Claude Sonnet 5.5","cache-break-even-reads/1-hour write (2× input), whole prefix new","Reuses before a cached prefix costs less, by model and write type (calculation) (1-hour write (2× input), whole prefix new)",1.11,"score","1.11","\u0001","\u0001","\u0001","calculation: Agent’s recorded tokens at this model’s list price",true,"\u0001","\u0001","\u0001"],["prompt-cache-break-even","cache-break-even-reads","1-hour write, pooled n = 6 session share, 19% already cached (as recorded)","Claude Sonnet 5.5","cache-break-even-reads/1-hour write, pooled n = 6 session share, 19% already cached (as recorded)","Reuses before a cached prefix costs less, by model and write type (calculation) (1-hour write, pooled n = 6 session share, 19% already cached (as recorded))",0.72,"score","0.72","\u0001","\u0001","\u0001","calculation: Agent’s recorded tokens at this model’s list price",true,"\u0001","\u0001","\u0001"],["prompt-cache-break-even","cache-break-even-reads","5-minute write (1.25× input, an assumption)","Claude Sonnet 5.5","cache-break-even-reads/5-minute write (1.25× input, an assumption)","Reuses before a cached prefix costs less, by model and write type (calculation) (5-minute write (1.25× input, an assumption))",0.28,"score","0.28","\u0001","\u0001","\u0001","calculation: Agent’s recorded tokens at this model’s list price",true,"\u0001","\u0001","\u0001"],["prompt-cache-break-even","cache-break-even-cost-curve","Claude Sonnet 5.5 (no cache)","1 turn","cache-break-even-cost-curve/1 turn","Cost of a reused prefix with and without the cache, by session length (calculation): 1 turn",15.66,"usd","$15.66","\u0001","\u0001","\u0001","no cache",true,"\u0001","\u0001","\u0001"],["prompt-cache-break-even","cache-break-even-cost-curve","Claude Sonnet 5.5 (no cache)","2 turns","cache-break-even-cost-curve/2 turns","Cost of a reused prefix with and without the cache, by session length (calculation): 2 turns",31.32,"usd","$31.32","\u0001","\u0001","\u0001","no cache",true,"\u0001","\u0001","\u0001"],["prompt-cache-break-even","cache-break-even-cost-curve","Claude Sonnet 5.5 (no cache)","3 turns","cache-break-even-cost-curve/3 turns","Cost of a reused prefix with and without the cache, by session length (calculation): 3 turns",46.99,"usd","$46.99","\u0001","\u0001","\u0001","no cache",true,"\u0001","\u0001","\u0001"],["prompt-cache-break-even","cache-break-even-cost-curve","Claude Sonnet 5.5 (no cache)","5 turns","cache-break-even-cost-curve/5 turns","Cost of a reused prefix with and without the cache, by session length (calculation): 5 turns",78.31,"usd","$78.31","\u0001","\u0001","\u0001","no cache",true,"\u0001","\u0001","\u0001"],["prompt-cache-break-even","cache-break-even-cost-curve","Claude Sonnet 5.5 (no cache)","10 turns","cache-break-even-cost-curve/10 turns","Cost of a reused prefix with and without the cache, by session length (calculation): 10 turns",156.62,"usd","$156.62","\u0001","\u0001","\u0001","no cache",true,"\u0001","\u0001","\u0001"],["prompt-cache-break-even","cache-break-even-cost-curve","Claude Sonnet 5.5 (no cache)","20 turns","cache-break-even-cost-curve/20 turns","Cost of a reused prefix with and without the cache, by session length (calculation): 20 turns",313.24,"usd","$313.24","\u0001","\u0001","\u0001","no cache",true,"\u0001","\u0001","\u0001"],["prompt-cache-break-even","cache-break-even-cost-curve","Claude Sonnet 5.5 (1-hour cache write)","1 turn","cache-break-even-cost-curve/1 turn","Cost of a reused prefix with and without the cache, by session length (calculation): 1 turn",31.32,"usd","$31.32","\u0001","\u0001","\u0001","1-hour cache write",true,"\u0001","\u0001","\u0001"],["prompt-cache-break-even","cache-break-even-cost-curve","Claude Sonnet 5.5 (1-hour cache write)","2 turns","cache-break-even-cost-curve/2 turns","Cost of a reused prefix with and without the cache, by session length (calculation): 2 turns",32.89,"usd","$32.89","\u0001","\u0001","\u0001","1-hour cache write",true,"\u0001","\u0001","\u0001"],["prompt-cache-break-even","cache-break-even-cost-curve","Claude Sonnet 5.5 (1-hour cache write)","3 turns","cache-break-even-cost-curve/3 turns","Cost of a reused prefix with and without the cache, by session length (calculation): 3 turns",34.46,"usd","$34.46","\u0001","\u0001","\u0001","1-hour cache write",true,"\u0001","\u0001","\u0001"],["prompt-cache-break-even","cache-break-even-cost-curve","Claude Sonnet 5.5 (1-hour cache write)","5 turns","cache-break-even-cost-curve/5 turns","Cost of a reused prefix with and without the cache, by session length (calculation): 5 turns",37.59,"usd","$37.59","\u0001","\u0001","\u0001","1-hour cache write",true,"\u0001","\u0001","\u0001"],["prompt-cache-break-even","cache-break-even-cost-curve","Claude Sonnet 5.5 (1-hour cache write)","10 turns","cache-break-even-cost-curve/10 turns","Cost of a reused prefix with and without the cache, by session length (calculation): 10 turns",45.42,"usd","$45.42","\u0001","\u0001","\u0001","1-hour cache write",true,"\u0001","\u0001","\u0001"],["prompt-cache-break-even","cache-break-even-cost-curve","Claude Sonnet 5.5 (1-hour cache write)","20 turns","cache-break-even-cost-curve/20 turns","Cost of a reused prefix with and without the cache, by session length (calculation): 20 turns",61.08,"usd","$61.08","\u0001","\u0001","\u0001","1-hour cache write",true,"\u0001","\u0001","\u0001"],["prompt-cache-break-even","cache-break-even-session-split","No cache (the same either way)","Claude Sonnet 5.5","cache-break-even-session-split/No cache (the same either way)","One 10-turn session or ten 1-turn sessions: cost with and without the cache (calculation) (No cache (the same either way))",156.62,"usd","$156.62","\u0001","\u0001","\u0001","",true,"\u0001","\u0001","\u0001"],["prompt-cache-break-even","cache-break-even-session-split","One 10-turn session, 1-hour cache","Claude Sonnet 5.5","cache-break-even-session-split/One 10-turn session, 1-hour cache","One 10-turn session or ten 1-turn sessions: cost with and without the cache (calculation) (One 10-turn session, 1-hour cache)",45.42,"usd","$45.42","\u0001","\u0001","\u0001","",true,"\u0001","\u0001","\u0001"],["prompt-cache-break-even","cache-break-even-session-split","Ten 1-turn sessions, 1-hour cache (assumed no reuse of the new prefix)","Claude Sonnet 5.5","cache-break-even-session-split/Ten 1-turn sessions, 1-hour cache (assumed no reuse of the new prefix)","One 10-turn session or ten 1-turn sessions: cost with and without the cache (calculation) (Ten 1-turn sessions, 1-hour cache (assumed no reuse of the new prefix))",313.24,"usd","$313.24","\u0001","\u0001","\u0001","",true,"\u0001","\u0001","\u0001"],["routing-holdout","routing-holdout-exact","Exact rate","Claude Sonnet 5.5 (low) · Claude Code","routing-holdout-exact","Unseen routing decisions answered exactly right",0.875,"rate","88% (49/56)",56,[0.7637,0.9381],"ci95","Claude Code · effort low","\u0001","\u0001","higher","\u0001"],["routing-holdout","routing-holdout-key-accuracy","Key accuracy","Claude Sonnet 5.5 (low) · Claude Code","routing-holdout-key-accuracy","Per-question accuracy on unseen decisions",0.92,"rate","92% (115/125)",125,[0.859,0.956],"ci95","Claude Code · effort low","\u0001","\u0001","higher","\u0001"],["routing-holdout","routing-holdout-by-purpose","Claude Sonnet 5.5 (low) · Claude Code","Failure class","routing-holdout-by-purpose/Failure class","Exact rate on unseen decisions, by decision type: Failure class",1,"rate","100% (14/14)",14,[0.7847,1],"ci95","Claude Code · effort low","\u0001","\u0001","higher","\u0001"],["routing-holdout","routing-holdout-by-purpose","Claude Sonnet 5.5 (low) · Claude Code","Message intent","routing-holdout-by-purpose/Message intent","Exact rate on unseen decisions, by decision type: Message intent",1,"rate","100% (14/14)",14,[0.7847,1],"ci95","Claude Code · effort low","\u0001","\u0001","higher","\u0001"],["routing-holdout","routing-holdout-by-purpose","Claude Sonnet 5.5 (low) · Claude Code","Is it a rule?","routing-holdout-by-purpose/Is it a rule?","Exact rate on unseen decisions, by decision type: Is it a rule?",0.9286,"rate","93% (13/14)",14,[0.6853,0.9873],"ci95","Claude Code · effort low","\u0001","\u0001","higher","\u0001"],["routing-holdout","routing-holdout-by-purpose","Claude Sonnet 5.5 (low) · Claude Code","Context shape","routing-holdout-by-purpose/Context shape","Exact rate on unseen decisions, by decision type: Context shape",0.5714,"rate","57% (8/14)",14,[0.3259,0.7862],"ci95","Claude Code · effort low","\u0001","\u0001","higher","\u0001"],["routing-holdout","routing-holdout-tuned-vs-unseen","Tuned set (routing-jev-vs-llm)","Claude Sonnet 5.5 (low) · Claude Code","routing-holdout-tuned-vs-unseen/Tuned set (routing-jev-vs-llm)","Tuned case set vs unseen holdout: exact rate per router (Tuned set (routing-jev-vs-llm))",0.939,"rate","94% (77/82)",82,[0.8651,0.9737],"ci95","Claude Code · effort low","\u0001","\u0001","higher","\u0001"],["routing-holdout","routing-holdout-tuned-vs-unseen","Unseen holdout","Claude Sonnet 5.5 (low) · Claude Code","routing-holdout-tuned-vs-unseen/Unseen holdout","Tuned case set vs unseen holdout: exact rate per router (Unseen holdout)",0.875,"rate","88% (49/56)",56,[0.7637,0.9381],"ci95","Claude Code · effort low","\u0001","\u0001","higher","\u0001"],["routing-holdout","routing-holdout-latency","Wall time","Claude Sonnet 5.5 (low) · Claude Code","routing-holdout-latency/Wall time","Time per routing decision, by route (Wall time)",2.359,"seconds","2.36 s",56,"\u0001","p50-p95","Claude Code · effort low","\u0001",[2.359,3.657],"\u0001","\u0001"],["routing-holdout","routing-holdout-latency","Model time (API, CLI-reported)","Claude Sonnet 5.5 (low) · Claude Code","routing-holdout-latency/Model time (API, CLI-reported)","Time per routing decision, by route (Model time (API, CLI-reported))",1.485,"seconds","1.49 s",56,"\u0001","p50-p95","Claude Code · effort low","\u0001",[1.485,2.377],"\u0001","\u0001"],["routing-holdout","routing-holdout-cost-per-1000","Cost","Claude Sonnet 5.5 (low) · Claude Code","routing-holdout-cost-per-1000","Cost per 1,000 unseen routing decisions",7.244,"usd","$7.24",56,"\u0001","\u0001","Claude Code · effort low",true,"\u0001","\u0001","\u0001"],["routing-holdout","\u0001","\u0001","\u0001","stat:holdout-gap-claude-sonnet","Claude Sonnet 5.5 (low) · Claude Code: holdout minus tuned-set exact rate",-0.064,"rate","−6.4 points",56,"\u0001","\u0001","low",true,"\u0001","\u0001","holdout-gap-claude-sonnet"],["thinking-token-bill","thinking-bill-share","Median call","Claude Sonnet 5.5 · Claude Code","thinking-bill-share","Reasoning share of output tokens per call on hard tasks (calculation)",54.54,"percent","54.5%",24,"\u0001","minmax","Claude Code",true,[0,95.91],"none","\u0001"],["thinking-token-bill","thinking-bill-cost-per-call","Reasoning (output tokens)","Claude Sonnet 5.5 · Claude Code","thinking-bill-cost-per-call/Reasoning (output tokens)","List-price cost per call: reasoning, remaining output and input (calculation) (Reasoning (output tokens))",0.006665,"usd","$0.0067",24,"\u0001","\u0001","Claude Code",true,"\u0001","none","\u0001"],["thinking-token-bill","thinking-bill-cost-per-call","Remaining output (visible-answer estimate)","Claude Sonnet 5.5 · Claude Code","thinking-bill-cost-per-call/Remaining output (visible-answer estimate)","List-price cost per call: reasoning, remaining output and input (calculation) (Remaining output (visible-answer estimate))",0.003672,"usd","$0.0037",24,"\u0001","\u0001","Claude Code",true,"\u0001","none","\u0001"],["thinking-token-bill","thinking-bill-cost-per-call","Input (prompt, cache priced)","Claude Sonnet 5.5 · Claude Code","thinking-bill-cost-per-call/Input (prompt, cache priced)","List-price cost per call: reasoning, remaining output and input (calculation) (Input (prompt, cache priced))",0.004012,"usd","$0.0040",24,"\u0001","\u0001","Claude Code",true,"\u0001","none","\u0001"],["thinking-token-bill","thinking-bill-by-effort","Reasoning cost per strict pass","Claude Sonnet 5.5 (low) · Claude Code","thinking-bill-by-effort/Reasoning cost per strict pass","Reasoning cost per strict pass by effort, with the total (calculation) (Reasoning cost per strict pass)",0.004317,"usd","$0.0043",16,"\u0001","\u0001","Claude Code · effort low",true,"\u0001","none","\u0001"],["thinking-token-bill","thinking-bill-by-effort","Reasoning cost per strict pass","Claude Sonnet 5.5 (medium) · Claude Code","thinking-bill-by-effort/Reasoning cost per strict pass","Reasoning cost per strict pass by effort, with the total (calculation) (Reasoning cost per strict pass)",0.005946,"usd","$0.0059",16,"\u0001","\u0001","Claude Code · effort medium",true,"\u0001","none","\u0001"],["thinking-token-bill","thinking-bill-by-effort","Reasoning cost per strict pass","Claude Sonnet 5.5 (high) · Claude Code","thinking-bill-by-effort/Reasoning cost per strict pass","Reasoning cost per strict pass by effort, with the total (calculation) (Reasoning cost per strict pass)",0.009369,"usd","$0.0094",16,"\u0001","\u0001","Claude Code · effort high",true,"\u0001","none","\u0001"],["thinking-token-bill","thinking-bill-by-effort","Reasoning cost per strict pass","Claude Sonnet 5.5 · Claude Code","thinking-bill-by-effort/Reasoning cost per strict pass","Reasoning cost per strict pass by effort, with the total (calculation) (Reasoning cost per strict pass)",0.006299,"usd","$0.0063",16,"\u0001","\u0001","Claude Code",true,"\u0001","none","\u0001"],["thinking-token-bill","thinking-bill-by-effort","Total cost per strict pass","Claude Sonnet 5.5 (low) · Claude Code","thinking-bill-by-effort/Total cost per strict pass","Reasoning cost per strict pass by effort, with the total (calculation) (Total cost per strict pass)",0.012191,"usd","$0.012",16,"\u0001","\u0001","Claude Code · effort low",true,"\u0001","none","\u0001"],["thinking-token-bill","thinking-bill-by-effort","Total cost per strict pass","Claude Sonnet 5.5 (medium) · Claude Code","thinking-bill-by-effort/Total cost per strict pass","Reasoning cost per strict pass by effort, with the total (calculation) (Total cost per strict pass)",0.01352,"usd","$0.014",16,"\u0001","\u0001","Claude Code · effort medium",true,"\u0001","none","\u0001"],["thinking-token-bill","thinking-bill-by-effort","Total cost per strict pass","Claude Sonnet 5.5 (high) · Claude Code","thinking-bill-by-effort/Total cost per strict pass","Reasoning cost per strict pass by effort, with the total (calculation) (Total cost per strict pass)",0.016705,"usd","$0.017",16,"\u0001","\u0001","Claude Code · effort high",true,"\u0001","none","\u0001"],["thinking-token-bill","thinking-bill-by-effort","Total cost per strict pass","Claude Sonnet 5.5 · Claude Code","thinking-bill-by-effort/Total cost per strict pass","Reasoning cost per strict pass by effort, with the total (calculation) (Total cost per strict pass)",0.013978,"usd","$0.014",16,"\u0001","\u0001","Claude Code",true,"\u0001","none","\u0001"],["thinking-token-bill","thinking-bill-short-vs-hard","Eight hard tasks","Claude Sonnet 5.5 · Claude Code","thinking-bill-short-vs-hard/Eight hard tasks","Reasoning share on short tasks vs hard tasks (calculation) (Eight hard tasks)",54.54,"percent","54.5%",24,"\u0001","minmax","Claude Code",true,[0,95.91],"none","\u0001"],["thinking-token-bill","thinking-bill-short-vs-hard","Five short tasks","Claude Sonnet 5.5 · Claude Code","thinking-bill-short-vs-hard/Five short tasks","Reasoning share on short tasks vs hard tasks (calculation) (Five short tasks)",0,"percent","0%",15,"\u0001","minmax","Claude Code",true,[0,72.75],"none","\u0001"],["llm-speed-anatomy","speed-anatomy-first-text","Time to first text","Claude Sonnet 5.5 · Claude Code","speed-anatomy-first-text","Time to first text: a 250-line answer, six models",1.96,"seconds","1.96 s",4,"\u0001","minmax","Claude Code","\u0001",[0.88,4.09],"\u0001","\u0001"],["llm-speed-anatomy","speed-anatomy-output-speed","Visible tokens per second","Claude Sonnet 5.5 · Claude Code","speed-anatomy-output-speed","Output speed after the first text: visible tokens per second (calculation)",231.7,"tokens","232",4,"\u0001","minmax","Claude Code",true,[230.3,233],"\u0001","\u0001"],["llm-speed-anatomy","speed-anatomy-chars-per-second","Characters per second","Claude Sonnet 5.5 · Claude Code","speed-anatomy-chars-per-second","Output speed in characters per second after the first text (calculation)",517,"count","517",4,"\u0001","minmax","Claude Code",true,[513,519],"\u0001","\u0001"],["llm-speed-anatomy","speed-anatomy-prompt-size","Claude Sonnet 5.5 · Claude Code","1k","speed-anatomy-prompt-size/1k","Time to first text as the prompt grows: 1k",1.45,"seconds","1.45 s",3,"\u0001","minmax","Claude Code",true,[1.23,1.72],"\u0001","\u0001"],["llm-speed-anatomy","speed-anatomy-prompt-size","Claude Sonnet 5.5 · Claude Code","16k","speed-anatomy-prompt-size/16k","Time to first text as the prompt grows: 16k",1.78,"seconds","1.78 s",3,"\u0001","minmax","Claude Code",true,[1.64,2.11],"\u0001","\u0001"],["llm-speed-anatomy","speed-anatomy-prompt-size","Claude Sonnet 5.5 · Claude Code","64k","speed-anatomy-prompt-size/64k","Time to first text as the prompt grows: 64k",3.07,"seconds","3.07 s",3,"\u0001","minmax","Claude Code",true,[1.38,3.61],"\u0001","\u0001"],["llm-speed-anatomy","speed-anatomy-total-by-size","1k prompt","Claude Sonnet 5.5 · Claude Code","speed-anatomy-total-by-size/1k prompt","Total time per call by prompt size (1k prompt)",1.78,"seconds","1.78 s",3,"\u0001","minmax","Claude Code","\u0001",[1.57,2.12],"\u0001","\u0001"],["llm-speed-anatomy","speed-anatomy-total-by-size","16k prompt","Claude Sonnet 5.5 · Claude Code","speed-anatomy-total-by-size/16k prompt","Total time per call by prompt size (16k prompt)",2.1,"seconds","2.10 s",3,"\u0001","minmax","Claude Code","\u0001",[1.98,2.48],"\u0001","\u0001"],["llm-speed-anatomy","speed-anatomy-total-by-size","64k prompt","Claude Sonnet 5.5 · Claude Code","speed-anatomy-total-by-size/64k prompt","Total time per call by prompt size (64k prompt)",3.44,"seconds","3.44 s",3,"\u0001","minmax","Claude Code","\u0001",[1.74,4.38],"\u0001","\u0001"],["llm-speed-anatomy","speed-anatomy-lookup-correct","Exact answer","Claude Sonnet 5.5 · Claude Code","speed-anatomy-lookup-correct","Exact lookup answers at the 1k, 16k and 64k prompt-size targets",1,"rate","100% (9/9)",9,[0.7009,1],"ci95","Claude Code","\u0001","\u0001","higher","\u0001"],["haiku-retry-or-escalate","retry-escalate-call-cost-by-task","Claude Sonnet 5.5 · Claude Code","Interval merge fix","retry-escalate-call-cost-by-task/Interval merge fix","List-price cost of one call, Haiku 4.5 vs Sonnet 5.5, on each hard task (calculation): Interval merge fix",0.00557,"usd","$0.0056",3,"\u0001","\u0001","Claude Code",true,"\u0001","\u0001","\u0001"],["haiku-retry-or-escalate","retry-escalate-call-cost-by-task","Claude Sonnet 5.5 · Claude Code","DST day length","retry-escalate-call-cost-by-task/DST day length","List-price cost of one call, Haiku 4.5 vs Sonnet 5.5, on each hard task (calculation): DST day length",0.02532,"usd","$0.025",3,"\u0001","\u0001","Claude Code",true,"\u0001","\u0001","\u0001"],["haiku-retry-or-escalate","retry-escalate-call-cost-by-task","Claude Sonnet 5.5 · Claude Code","CSV parser","retry-escalate-call-cost-by-task/CSV parser","List-price cost of one call, Haiku 4.5 vs Sonnet 5.5, on each hard task (calculation): CSV parser",0.01464,"usd","$0.015",3,"\u0001","\u0001","Claude Code",true,"\u0001","\u0001","\u0001"],["haiku-retry-or-escalate","retry-escalate-call-cost-by-task","Claude Sonnet 5.5 · Claude Code","Event-loop order","retry-escalate-call-cost-by-task/Event-loop order","List-price cost of one call, Haiku 4.5 vs Sonnet 5.5, on each hard task (calculation): Event-loop order",0.01588,"usd","$0.016",3,"\u0001","\u0001","Claude Code",true,"\u0001","\u0001","\u0001"],["haiku-retry-or-escalate","retry-escalate-call-cost-by-task","Claude Sonnet 5.5 · Claude Code","Room schedule","retry-escalate-call-cost-by-task/Room schedule","List-price cost of one call, Haiku 4.5 vs Sonnet 5.5, on each hard task (calculation): Room schedule",0.01243,"usd","$0.012",3,"\u0001","\u0001","Claude Code",true,"\u0001","\u0001","\u0001"],["haiku-retry-or-escalate","retry-escalate-call-cost-by-task","Claude Sonnet 5.5 · Claude Code","SemVer regex","retry-escalate-call-cost-by-task/SemVer regex","List-price cost of one call, Haiku 4.5 vs Sonnet 5.5, on each hard task (calculation): SemVer regex",0.00514,"usd","$0.0051",3,"\u0001","\u0001","Claude Code",true,"\u0001","\u0001","\u0001"],["haiku-retry-or-escalate","retry-escalate-call-cost-by-task","Claude Sonnet 5.5 · Claude Code","Money refactor","retry-escalate-call-cost-by-task/Money refactor","List-price cost of one call, Haiku 4.5 vs Sonnet 5.5, on each hard task (calculation): Money refactor",0.00961,"usd","$0.0096",3,"\u0001","\u0001","Claude Code",true,"\u0001","\u0001","\u0001"],["haiku-retry-or-escalate","retry-escalate-call-cost-by-task","Claude Sonnet 5.5 · Claude Code","SQLite report query","retry-escalate-call-cost-by-task/SQLite report query","List-price cost of one call, Haiku 4.5 vs Sonnet 5.5, on each hard task (calculation): SQLite report query",0.01719,"usd","$0.017",3,"\u0001","\u0001","Claude Code",true,"\u0001","\u0001","\u0001"],["harder-tasks-head-to-head","harder-h2h-pass-rate","Strict pass","Claude Sonnet 5.5 · Claude Code","harder-h2h-pass-rate/Strict pass","Pass rate on 4 harder tasks (Strict pass)",0.375,"rate","38% (6/16)",16,[0.1848,0.6136],"ci95","Claude Code","\u0001","\u0001","higher","\u0001"],["harder-tasks-head-to-head","harder-h2h-pass-rate","Lenient (format misses counted)","Claude Sonnet 5.5 · Claude Code","harder-h2h-pass-rate/Lenient (format misses counted)","Pass rate on 4 harder tasks (Lenient (format misses counted))",0.375,"rate","38% (6/16)",16,[0.1848,0.6136],"ci95","Claude Code","\u0001","\u0001","higher","\u0001"],["harder-tasks-head-to-head","harder-h2h-tool-attempts","Tool attempt","Claude Sonnet 5.5 · Claude Code","harder-h2h-tool-attempts","Calls that tried a tool although tools were off",0.3125,"rate","31% (5/16)",16,[0.1416,0.556],"ci95","Claude Code","\u0001","\u0001","none","\u0001"],["harder-tasks-head-to-head","harder-h2h-pass-by-task","Claude Sonnet 5.5 · Claude Code","10x10 nonogram","harder-h2h-pass-by-task/10x10 nonogram","Strict pass rate by task: 10x10 nonogram",1,"rate","100% (4/4)",4,[0.5101,1],"ci95","Claude Code","\u0001","\u0001","higher","\u0001"],["harder-tasks-head-to-head","harder-h2h-pass-by-task","Claude Sonnet 5.5 · Claude Code","Sudoku, 22 givens","harder-h2h-pass-by-task/Sudoku, 22 givens","Strict pass rate by task: Sudoku, 22 givens",0,"rate","0% (0/4)",4,[0,0.4899],"ci95","Claude Code","\u0001","\u0001","higher","\u0001"],["harder-tasks-head-to-head","harder-h2h-pass-by-task","Claude Sonnet 5.5 · Claude Code","6x6 Skyscrapers","harder-h2h-pass-by-task/6x6 Skyscrapers","Strict pass rate by task: 6x6 Skyscrapers",0,"rate","0% (0/4)",4,[0,0.4899],"ci95","Claude Code","\u0001","\u0001","higher","\u0001"],["harder-tasks-head-to-head","harder-h2h-pass-by-task","Claude Sonnet 5.5 · Claude Code","Seeded shuffle output","harder-h2h-pass-by-task/Seeded shuffle output","Strict pass rate by task: Seeded shuffle output",0.5,"rate","50% (2/4)",4,[0.15,0.85],"ci95","Claude Code","\u0001","\u0001","higher","\u0001"],["harder-tasks-head-to-head","harder-h2h-total-latency","Total time per call","Claude Sonnet 5.5 · Claude Code","harder-h2h-total-latency","Total time per call on harder tasks",70.43,"seconds","70.4 s",12,"\u0001","minmax","Claude Code","\u0001",[4.32,210.08],"\u0001","\u0001"],["harder-tasks-head-to-head","harder-h2h-output-tokens","Output tokens","Claude Sonnet 5.5 · Claude Code","harder-h2h-output-tokens/Output tokens","Output tokens per call on harder tasks (Output tokens)",9287,"tokens","9,287",12,"\u0001","minmax","Claude Code","\u0001",[407,27921],"none","\u0001"],["harder-tasks-head-to-head","harder-h2h-cost-per-pass","Cost per strict pass","Claude Sonnet 5.5 · Claude Code","harder-h2h-cost-per-pass","List-price cost per strict pass on harder tasks (calculation)",0.23843,"usd","$0.24",16,"\u0001","\u0001","Claude Code",true,"\u0001","\u0001","\u0001"]]}],["claude-opus-5-5","Claude Opus 5.5","Anthropic","model","Anthropic’s large Claude model, measured through Claude Code at low, medium, high and default effort.",["Claude Opus 5.5","Opus 5.5"],{"$k":["studySlug","chartId","series","point","metric","label","value","unit","display","n","ci","spanKind","context","calculation","range","polarity"],"$r":[["swe-bench-opus-vs-sonnet","swebench-opus-sonnet-resolved","Resolved","Claude Opus 5.5 (Agent, new build)","swebench-opus-sonnet-resolved","Resolved on the same 3 SWE-bench Verified instances (interim)",0.6667,"rate","67% (2/3)",3,[0.2077,0.9385],"ci95","Agent · new build · SWE-bench Verified, interim paired probe","\u0001","\u0001","\u0001"],["swe-bench-opus-vs-sonnet","swebench-opus-sonnet-cost-per-attempt","List-price cost per attempt","Claude Opus 5.5 (Agent, new build)","swebench-opus-sonnet-cost-per-attempt","List-price cost per attempt (calculation)",7.59,"usd","$7.59",3,"\u0001","\u0001","Agent · new build · SWE-bench Verified, interim paired probe",true,"\u0001","\u0001"],["swe-bench-opus-vs-sonnet","swebench-opus-sonnet-minutes","Worker minutes per attempt","Claude Opus 5.5 (Agent, new build)","swebench-opus-sonnet-minutes","Worker time per attempt",20.26,"minutes","20.3 min",3,"\u0001","minmax","Agent · new build · SWE-bench Verified, interim paired probe","\u0001",[9.76,25.29],"\u0001"],["model-head-to-head","h2h-pass-rate","Pass rate","Claude Opus 5.5 (high) · Claude Code","h2h-pass-rate","Pass rate on five validated tasks",1,"rate","100% (15/15)",15,[0.7961,1],"ci95","Claude Code · effort high · five short validated tasks","\u0001","\u0001","\u0001"],["model-head-to-head","h2h-pass-rate","Pass rate","Claude Opus 5.5 · Claude Code","h2h-pass-rate","Pass rate on five validated tasks",1,"rate","100% (15/15)",15,[0.7961,1],"ci95","Claude Code · five short validated tasks","\u0001","\u0001","\u0001"],["model-head-to-head","h2h-pass-rate","Pass rate","Claude Opus 5.5 (low) · Claude Code","h2h-pass-rate","Pass rate on five validated tasks",1,"rate","100% (15/15)",15,[0.7961,1],"ci95","Claude Code · effort low · five short validated tasks","\u0001","\u0001","\u0001"],["model-head-to-head","h2h-total-latency","Total time per call","Claude Opus 5.5 (high) · Claude Code","h2h-total-latency","Total time per call",2.71,"seconds","2.71 s",15,"\u0001","minmax","Claude Code · effort high · five short validated tasks","\u0001",[2.45,11.78],"\u0001"],["model-head-to-head","h2h-total-latency","Total time per call","Claude Opus 5.5 · Claude Code","h2h-total-latency","Total time per call",2.75,"seconds","2.75 s",15,"\u0001","minmax","Claude Code · five short validated tasks","\u0001",[2.47,8.91],"\u0001"],["model-head-to-head","h2h-total-latency","Total time per call","Claude Opus 5.5 (low) · Claude Code","h2h-total-latency","Total time per call",2.83,"seconds","2.83 s",15,"\u0001","minmax","Claude Code · effort low · five short validated tasks","\u0001",[2.35,6.62],"\u0001"],["model-head-to-head","h2h-first-useful-latency","Time to first useful output","Claude Opus 5.5 (high) · Claude Code","h2h-first-useful-latency","Time to first useful output",2.04,"seconds","2.04 s",15,"\u0001","minmax","Claude Code · effort high · five short validated tasks","\u0001",[1.4,9.94],"\u0001"],["model-head-to-head","h2h-first-useful-latency","Time to first useful output","Claude Opus 5.5 · Claude Code","h2h-first-useful-latency","Time to first useful output",1.92,"seconds","1.92 s",15,"\u0001","minmax","Claude Code · five short validated tasks","\u0001",[1.56,7.23],"\u0001"],["model-head-to-head","h2h-first-useful-latency","Time to first useful output","Claude Opus 5.5 (low) · Claude Code","h2h-first-useful-latency","Time to first useful output",2.39,"seconds","2.39 s",15,"\u0001","minmax","Claude Code · effort low · five short validated tasks","\u0001",[1.45,4.9],"\u0001"],["model-head-to-head","h2h-input-tokens","Cache read","Claude Opus 5.5 (high) · Claude Code","h2h-input-tokens/Cache read","Input tokens per call: what the CLI sends (Cache read)",1463,"tokens","1,463",15,"\u0001","\u0001","Claude Code · effort high · five short validated tasks","\u0001","\u0001","\u0001"],["model-head-to-head","h2h-input-tokens","Cache read","Claude Opus 5.5 · Claude Code","h2h-input-tokens/Cache read","Input tokens per call: what the CLI sends (Cache read)",1401,"tokens","1,401",15,"\u0001","\u0001","Claude Code · five short validated tasks","\u0001","\u0001","\u0001"],["model-head-to-head","h2h-input-tokens","Cache read","Claude Opus 5.5 (low) · Claude Code","h2h-input-tokens/Cache read","Input tokens per call: what the CLI sends (Cache read)",1463,"tokens","1,463",15,"\u0001","\u0001","Claude Code · effort low · five short validated tasks","\u0001","\u0001","\u0001"],["model-head-to-head","h2h-input-tokens","Other input","Claude Opus 5.5 (high) · Claude Code","h2h-input-tokens/Other input","Input tokens per call: what the CLI sends (Other input)",619,"tokens","619",15,"\u0001","\u0001","Claude Code · effort high · five short validated tasks","\u0001","\u0001","\u0001"],["model-head-to-head","h2h-input-tokens","Other input","Claude Opus 5.5 · Claude Code","h2h-input-tokens/Other input","Input tokens per call: what the CLI sends (Other input)",680,"tokens","680",15,"\u0001","\u0001","Claude Code · five short validated tasks","\u0001","\u0001","\u0001"],["model-head-to-head","h2h-input-tokens","Other input","Claude Opus 5.5 (low) · Claude Code","h2h-input-tokens/Other input","Input tokens per call: what the CLI sends (Other input)",618,"tokens","618",15,"\u0001","\u0001","Claude Code · effort low · five short validated tasks","\u0001","\u0001","\u0001"],["model-head-to-head","h2h-output-tokens","Output tokens","Claude Opus 5.5 (high) · Claude Code","h2h-output-tokens/Output tokens","Output tokens per call (Output tokens)",78,"tokens","78",15,"\u0001","\u0001","Claude Code · effort high · five short validated tasks","\u0001","\u0001","\u0001"],["model-head-to-head","h2h-output-tokens","Output tokens","Claude Opus 5.5 · Claude Code","h2h-output-tokens/Output tokens","Output tokens per call (Output tokens)",64,"tokens","64",15,"\u0001","\u0001","Claude Code · five short validated tasks","\u0001","\u0001","\u0001"],["model-head-to-head","h2h-output-tokens","Output tokens","Claude Opus 5.5 (low) · Claude Code","h2h-output-tokens/Output tokens","Output tokens per call (Output tokens)",64,"tokens","64",15,"\u0001","\u0001","Claude Code · effort low · five short validated tasks","\u0001","\u0001","\u0001"],["model-head-to-head","h2h-list-price-per-call","Cost per call","Claude Opus 5.5 (high) · Claude Code","h2h-list-price-per-call","List-price cost per call (calculation)",0.00694,"usd","$0.0069",15,"\u0001","minmax","Claude Code · effort high · five short validated tasks",true,[0.00592,0.02708],"\u0001"],["model-head-to-head","h2h-list-price-per-call","Cost per call","Claude Opus 5.5 · Claude Code","h2h-list-price-per-call","List-price cost per call (calculation)",0.00688,"usd","$0.0069",15,"\u0001","minmax","Claude Code · five short validated tasks",true,[0.00592,0.02226],"\u0001"],["model-head-to-head","h2h-list-price-per-call","Cost per call","Claude Opus 5.5 (low) · Claude Code","h2h-list-price-per-call","List-price cost per call (calculation)",0.00688,"usd","$0.0069",15,"\u0001","minmax","Claude Code · effort low · five short validated tasks",true,[0.00582,0.01793],"\u0001"],["model-head-to-head","h2h-cost-per-pass","Cost per pass","Claude Opus 5.5 (low) · Claude Code","h2h-cost-per-pass","List-price cost per passing answer (calculation)",0.00829,"usd","$0.0083",15,"\u0001","\u0001","Claude Code · effort low · five short validated tasks",true,"\u0001","\u0001"],["model-head-to-head","h2h-cost-per-pass","Cost per pass","Claude Opus 5.5 · Claude Code","h2h-cost-per-pass","List-price cost per passing answer (calculation)",0.01009,"usd","$0.010",15,"\u0001","\u0001","Claude Code · five short validated tasks",true,"\u0001","\u0001"],["model-head-to-head","h2h-cost-per-pass","Cost per pass","Claude Opus 5.5 (high) · Claude Code","h2h-cost-per-pass","List-price cost per passing answer (calculation)",0.01049,"usd","$0.010",15,"\u0001","\u0001","Claude Code · effort high · five short validated tasks",true,"\u0001","\u0001"],["hard-model-head-to-head","hard-h2h-pass-rate","Strict pass","Claude Opus 5.5 · Claude Code","hard-h2h-pass-rate/Strict pass","Pass rate on eight hard tasks (Strict pass)",1,"rate","100% (24/24)",24,[0.862,1],"ci95","Claude Code · eight hard validated tasks","\u0001","\u0001","\u0001"],["hard-model-head-to-head","hard-h2h-pass-rate","Strict pass","Claude Opus 5.5 (high) · Claude Code","hard-h2h-pass-rate/Strict pass","Pass rate on eight hard tasks (Strict pass)",1,"rate","100% (24/24)",24,[0.862,1],"ci95","Claude Code · effort high · eight hard validated tasks","\u0001","\u0001","\u0001"],["hard-model-head-to-head","hard-h2h-pass-rate","Lenient (format misses counted)","Claude Opus 5.5 · Claude Code","hard-h2h-pass-rate/Lenient (format misses counted)","Pass rate on eight hard tasks (Lenient (format misses counted))",1,"rate","100% (24/24)",24,[0.862,1],"ci95","Claude Code · eight hard validated tasks","\u0001","\u0001","\u0001"],["hard-model-head-to-head","hard-h2h-pass-rate","Lenient (format misses counted)","Claude Opus 5.5 (high) · Claude Code","hard-h2h-pass-rate/Lenient (format misses counted)","Pass rate on eight hard tasks (Lenient (format misses counted))",1,"rate","100% (24/24)",24,[0.862,1],"ci95","Claude Code · effort high · eight hard validated tasks","\u0001","\u0001","\u0001"],["hard-model-head-to-head","hard-h2h-total-latency","Total time per call on hard tasks (separate batches)","Claude Opus 5.5 · Claude Code","hard-h2h-total-latency","Total time per call on hard tasks (separate batches)",9.18,"seconds","9.18 s",24,"\u0001","minmax","Claude Code · eight hard validated tasks","\u0001",[4.24,27.21],"\u0001"],["hard-model-head-to-head","hard-h2h-total-latency","Total time per call on hard tasks (separate batches)","Claude Opus 5.5 (high) · Claude Code","hard-h2h-total-latency","Total time per call on hard tasks (separate batches)",11.03,"seconds","11.0 s",24,"\u0001","minmax","Claude Code · effort high · eight hard validated tasks","\u0001",[3.63,63],"\u0001"],["hard-model-head-to-head","hard-h2h-first-useful-latency","Time to first useful output on hard tasks","Claude Opus 5.5 · Claude Code","hard-h2h-first-useful-latency","Time to first useful output on hard tasks",6.78,"seconds","6.78 s",24,"\u0001","minmax","Claude Code · eight hard validated tasks","\u0001",[2.39,21.77],"\u0001"],["hard-model-head-to-head","hard-h2h-first-useful-latency","Time to first useful output on hard tasks","Claude Opus 5.5 (high) · Claude Code","hard-h2h-first-useful-latency","Time to first useful output on hard tasks",7.13,"seconds","7.13 s",24,"\u0001","minmax","Claude Code · effort high · eight hard validated tasks","\u0001",[2.15,56.23],"\u0001"],["hard-model-head-to-head","hard-h2h-output-tokens","Output tokens","Claude Opus 5.5 · Claude Code","hard-h2h-output-tokens/Output tokens","Output tokens per call on hard tasks (Output tokens)",945,"tokens","945",24,"\u0001","\u0001","Claude Code · eight hard validated tasks","\u0001","\u0001","\u0001"],["hard-model-head-to-head","hard-h2h-output-tokens","Output tokens","Claude Opus 5.5 (high) · Claude Code","hard-h2h-output-tokens/Output tokens","Output tokens per call on hard tasks (Output tokens)",1052,"tokens","1,052",24,"\u0001","\u0001","Claude Code · effort high · eight hard validated tasks","\u0001","\u0001","\u0001"],["hard-model-head-to-head","hard-h2h-cost-per-pass","Cost per strict pass","Claude Opus 5.5 · Claude Code","hard-h2h-cost-per-pass","List-price cost per strict pass on hard tasks (calculation)",0.02824,"usd","$0.028",24,"\u0001","\u0001","Claude Code · eight hard validated tasks",true,"\u0001","\u0001"],["hard-model-head-to-head","hard-h2h-cost-per-pass","Cost per strict pass","Claude Opus 5.5 (high) · Claude Code","hard-h2h-cost-per-pass","List-price cost per strict pass on hard tasks (calculation)",0.03337,"usd","$0.033",24,"\u0001","\u0001","Claude Code · effort high · eight hard validated tasks",true,"\u0001","\u0001"],["coding-agents-head-to-head","coding-agents-pass-rate","Passed every hidden check","Claude Opus 5.5 · Claude Code","coding-agents-pass-rate","Coding sessions that passed every hidden check",1,"rate","100% (12/12)",12,[0.7575,1],"ci95","Claude Code · six small repository tasks with hidden tests","\u0001","\u0001","\u0001"],["coding-agents-head-to-head","coding-agents-wall-time","Wall time per session","Claude Opus 5.5 · Claude Code","coding-agents-wall-time","Time per coding session",56.9,"seconds","56.9 s",12,"\u0001","minmax","Claude Code · six small repository tasks with hidden tests","\u0001",[29.8,185.8],"\u0001"],["coding-agents-head-to-head","coding-agents-tool-calls","Tool calls per session","Claude Opus 5.5 · Claude Code","coding-agents-tool-calls","Tool calls per coding session",7.5,"calls","7.5",12,"\u0001","minmax","Claude Code · six small repository tasks with hidden tests","\u0001",[5,14],"\u0001"],["coding-agents-head-to-head","coding-agents-cost-per-pass","List-price cost per pass","Claude Opus 5.5 · Claude Code","coding-agents-cost-per-pass","List-price cost per passing coding session (calculation)",0.2229,"usd","$0.22",12,"\u0001","\u0001","Claude Code · six small repository tasks with hidden tests",true,"\u0001","\u0001"],["effort-ladder","effort-ladder-pass-rate","Strict pass","Claude Opus 5.5 (low) · Claude Code","effort-ladder-pass-rate","Strict pass rate by effort on eight hard tasks",1,"rate","100% (16/16)",16,[0.8064,1],"ci95","Claude Code · effort low · eight hard validated tasks, effort ladder","\u0001","\u0001","\u0001"],["effort-ladder","effort-ladder-pass-rate","Strict pass","Claude Opus 5.5 (medium) · Claude Code","effort-ladder-pass-rate","Strict pass rate by effort on eight hard tasks",1,"rate","100% (16/16)",16,[0.8064,1],"ci95","Claude Code · effort medium · eight hard validated tasks, effort ladder","\u0001","\u0001","\u0001"],["effort-ladder","effort-ladder-pass-rate","Strict pass","Claude Opus 5.5 (high) · Claude Code","effort-ladder-pass-rate","Strict pass rate by effort on eight hard tasks",1,"rate","100% (16/16)",16,[0.8064,1],"ci95","Claude Code · effort high · eight hard validated tasks, effort ladder","\u0001","\u0001","\u0001"],["effort-ladder","effort-ladder-pass-rate","Strict pass","Claude Opus 5.5 · Claude Code","effort-ladder-pass-rate","Strict pass rate by effort on eight hard tasks",1,"rate","100% (16/16)",16,[0.8064,1],"ci95","Claude Code · eight hard validated tasks, effort ladder","\u0001","\u0001","\u0001"],["effort-ladder","effort-ladder-total-latency","Total time per call by effort on hard tasks","Claude Opus 5.5 (low) · Claude Code","effort-ladder-total-latency","Total time per call by effort on hard tasks",7.5,"seconds","7.50 s",16,"\u0001","minmax","Claude Code · effort low · eight hard validated tasks, effort ladder","\u0001",[3.34,15.82],"\u0001"],["effort-ladder","effort-ladder-total-latency","Total time per call by effort on hard tasks","Claude Opus 5.5 (medium) · Claude Code","effort-ladder-total-latency","Total time per call by effort on hard tasks",9.72,"seconds","9.72 s",16,"\u0001","minmax","Claude Code · effort medium · eight hard validated tasks, effort ladder","\u0001",[4.78,31.36],"\u0001"],["effort-ladder","effort-ladder-total-latency","Total time per call by effort on hard tasks","Claude Opus 5.5 (high) · Claude Code","effort-ladder-total-latency","Total time per call by effort on hard tasks",10.11,"seconds","10.1 s",16,"\u0001","minmax","Claude Code · effort high · eight hard validated tasks, effort ladder","\u0001",[3.63,63],"\u0001"],["effort-ladder","effort-ladder-total-latency","Total time per call by effort on hard tasks","Claude Opus 5.5 · Claude Code","effort-ladder-total-latency","Total time per call by effort on hard tasks",9.18,"seconds","9.18 s",16,"\u0001","minmax","Claude Code · eight hard validated tasks, effort ladder","\u0001",[4.24,27.21],"\u0001"],["effort-ladder","effort-ladder-output-tokens","Output tokens","Claude Opus 5.5 (low) · Claude Code","effort-ladder-output-tokens/Output tokens","Output tokens per call by effort on hard tasks (Output tokens)",594,"tokens","594",16,"\u0001","\u0001","Claude Code · effort low · eight hard validated tasks, effort ladder","\u0001","\u0001","\u0001"],["effort-ladder","effort-ladder-output-tokens","Output tokens","Claude Opus 5.5 (medium) · Claude Code","effort-ladder-output-tokens/Output tokens","Output tokens per call by effort on hard tasks (Output tokens)",853,"tokens","853",16,"\u0001","\u0001","Claude Code · effort medium · eight hard validated tasks, effort ladder","\u0001","\u0001","\u0001"],["effort-ladder","effort-ladder-output-tokens","Output tokens","Claude Opus 5.5 (high) · Claude Code","effort-ladder-output-tokens/Output tokens","Output tokens per call by effort on hard tasks (Output tokens)",1052,"tokens","1,052",16,"\u0001","\u0001","Claude Code · effort high · eight hard validated tasks, effort ladder","\u0001","\u0001","\u0001"],["effort-ladder","effort-ladder-output-tokens","Output tokens","Claude Opus 5.5 · Claude Code","effort-ladder-output-tokens/Output tokens","Output tokens per call by effort on hard tasks (Output tokens)",945,"tokens","945",16,"\u0001","\u0001","Claude Code · eight hard validated tasks, effort ladder","\u0001","\u0001","\u0001"],["effort-ladder","effort-ladder-cost-per-pass","Cost per strict pass","Claude Opus 5.5 (low) · Claude Code","effort-ladder-cost-per-pass","List-price cost per strict pass by effort (calculation)",0.02115,"usd","$0.021",16,"\u0001","\u0001","Claude Code · effort low · eight hard validated tasks, effort ladder",true,"\u0001","\u0001"],["effort-ladder","effort-ladder-cost-per-pass","Cost per strict pass","Claude Opus 5.5 (medium) · Claude Code","effort-ladder-cost-per-pass","List-price cost per strict pass by effort (calculation)",0.02947,"usd","$0.029",16,"\u0001","\u0001","Claude Code · effort medium · eight hard validated tasks, effort ladder",true,"\u0001","\u0001"],["effort-ladder","effort-ladder-cost-per-pass","Cost per strict pass","Claude Opus 5.5 (high) · Claude Code","effort-ladder-cost-per-pass","List-price cost per strict pass by effort (calculation)",0.03368,"usd","$0.034",16,"\u0001","\u0001","Claude Code · effort high · eight hard validated tasks, effort ladder",true,"\u0001","\u0001"],["effort-ladder","effort-ladder-cost-per-pass","Cost per strict pass","Claude Opus 5.5 · Claude Code","effort-ladder-cost-per-pass","List-price cost per strict pass by effort (calculation)",0.02893,"usd","$0.029",16,"\u0001","\u0001","Claude Code · eight hard validated tasks, effort ladder",true,"\u0001","\u0001"],["caching-consistency","caching-cost-with-without","With the cache, as recorded","Claude Opus 5.5 · Claude Code","caching-cost-with-without/With the cache, as recorded","List-price cost of 5-question sessions with and without the cache (calculation) (With the cache, as recorded)",0.255057,"usd","$0.26",15,"\u0001","\u0001","Claude Code · calculation: 5-turn cached sessions over a fixed ledger",true,"\u0001","\u0001"],["caching-consistency","caching-cost-with-without","Without a cache: every input token at the input price","Claude Opus 5.5 · Claude Code","caching-cost-with-without/Without a cache: every input token at the input price","List-price cost of 5-question sessions with and without the cache (calculation) (Without a cache: every input token at the input price)",0.544228,"usd","$0.54",15,"\u0001","\u0001","Claude Code · calculation: 5-turn cached sessions over a fixed ledger",true,"\u0001","\u0001"],["caching-consistency","caching-latency-first-vs-later","Turn 1 (writes the ledger to the cache)","Claude Opus 5.5 · Claude Code","caching-latency-first-vs-later/Turn 1 (writes the ledger to the cache)","Time per turn: first turn vs later turns in a cached session (Turn 1 (writes the ledger to the cache))",1.9,"seconds","1.90 s",3,"\u0001","minmax","Claude Code · 5-turn cached sessions over a fixed ledger","\u0001",[1.78,4.36],"\u0001"],["caching-consistency","caching-latency-first-vs-later","Turns 2-5 (read the ledger from the cache)","Claude Opus 5.5 · Claude Code","caching-latency-first-vs-later/Turns 2-5 (read the ledger from the cache)","Time per turn: first turn vs later turns in a cached session (Turns 2-5 (read the ledger from the cache))",2.4,"seconds","2.40 s",12,"\u0001","minmax","Claude Code · 5-turn cached sessions over a fixed ledger","\u0001",[1.63,12.67],"\u0001"],["cost-thought-experiments","repriced-cost-per-resolved","Repriced cost per resolved instance","Claude Opus 5.5","repriced-cost-per-resolved","Thought experiment: the same tokens at other list prices",5.753,"usd","$5.75","\u0001","\u0001","\u0001","calculation: Agent’s recorded tokens at this model’s list price",true,"\u0001","\u0001"],["cost-thought-experiments","prompt-cache-savings","With caching (as recorded)","Claude Opus 5.5","prompt-cache-savings/With caching (as recorded)","Thought experiment: what prompt caching saved (With caching (as recorded))",143.83,"usd","$143.83","\u0001","\u0001","\u0001","calculation: Agent’s recorded tokens at this model’s list price",true,"\u0001","\u0001"],["cost-thought-experiments","prompt-cache-savings","Without caching","Claude Opus 5.5","prompt-cache-savings/Without caching","Thought experiment: what prompt caching saved (Without caching)",686.66,"usd","$686.66","\u0001","\u0001","\u0001","calculation: Agent’s recorded tokens at this model’s list price",true,"\u0001","\u0001"],["prompt-cache-break-even","cache-break-even-reads","1-hour write (2× input), whole prefix new","Claude Opus 5.5 (cache read $0.2 per M)","cache-break-even-reads/1-hour write (2× input), whole prefix new","Reuses before a cached prefix costs less, by model and write type (calculation) (1-hour write (2× input), whole prefix new)",1.05,"score","1.05","\u0001","\u0001","\u0001","cache read $0.2 per M · calculation: Agent’s recorded tokens at this model’s list price",true,"\u0001","\u0001"],["prompt-cache-break-even","cache-break-even-reads","1-hour write (2× input), whole prefix new","Claude Opus 5.5 (cache read $0.4 per M)","cache-break-even-reads/1-hour write (2× input), whole prefix new","Reuses before a cached prefix costs less, by model and write type (calculation) (1-hour write (2× input), whole prefix new)",1.11,"score","1.11","\u0001","\u0001","\u0001","cache read $0.4 per M · calculation: Agent’s recorded tokens at this model’s list price",true,"\u0001","\u0001"],["prompt-cache-break-even","cache-break-even-reads","1-hour write, pooled n = 6 session share, 19% already cached (as recorded)","Claude Opus 5.5 (cache read $0.2 per M)","cache-break-even-reads/1-hour write, pooled n = 6 session share, 19% already cached (as recorded)","Reuses before a cached prefix costs less, by model and write type (calculation) (1-hour write, pooled n = 6 session share, 19% already cached (as recorded))",0.67,"score","0.67","\u0001","\u0001","\u0001","cache read $0.2 per M · calculation: Agent’s recorded tokens at this model’s list price",true,"\u0001","\u0001"],["prompt-cache-break-even","cache-break-even-reads","1-hour write, pooled n = 6 session share, 19% already cached (as recorded)","Claude Opus 5.5 (cache read $0.4 per M)","cache-break-even-reads/1-hour write, pooled n = 6 session share, 19% already cached (as recorded)","Reuses before a cached prefix costs less, by model and write type (calculation) (1-hour write, pooled n = 6 session share, 19% already cached (as recorded))",0.72,"score","0.72","\u0001","\u0001","\u0001","cache read $0.4 per M · calculation: Agent’s recorded tokens at this model’s list price",true,"\u0001","\u0001"],["prompt-cache-break-even","cache-break-even-reads","5-minute write (1.25× input, an assumption)","Claude Opus 5.5 (cache read $0.2 per M)","cache-break-even-reads/5-minute write (1.25× input, an assumption)","Reuses before a cached prefix costs less, by model and write type (calculation) (5-minute write (1.25× input, an assumption))",0.26,"score","0.26","\u0001","\u0001","\u0001","cache read $0.2 per M · calculation: Agent’s recorded tokens at this model’s list price",true,"\u0001","\u0001"],["prompt-cache-break-even","cache-break-even-reads","5-minute write (1.25× input, an assumption)","Claude Opus 5.5 (cache read $0.4 per M)","cache-break-even-reads/5-minute write (1.25× input, an assumption)","Reuses before a cached prefix costs less, by model and write type (calculation) (5-minute write (1.25× input, an assumption))",0.28,"score","0.28","\u0001","\u0001","\u0001","cache read $0.4 per M · calculation: Agent’s recorded tokens at this model’s list price",true,"\u0001","\u0001"],["prompt-cache-break-even","cache-break-even-cost-curve","Claude Opus 5.5 (no cache)","1 turn","cache-break-even-cost-curve/1 turn","Cost of a reused prefix with and without the cache, by session length (calculation): 1 turn",31.31,"usd","$31.31","\u0001","\u0001","\u0001","no cache",true,"\u0001","\u0001"],["prompt-cache-break-even","cache-break-even-cost-curve","Claude Opus 5.5 (no cache)","2 turns","cache-break-even-cost-curve/2 turns","Cost of a reused prefix with and without the cache, by session length (calculation): 2 turns",62.62,"usd","$62.62","\u0001","\u0001","\u0001","no cache",true,"\u0001","\u0001"],["prompt-cache-break-even","cache-break-even-cost-curve","Claude Opus 5.5 (no cache)","3 turns","cache-break-even-cost-curve/3 turns","Cost of a reused prefix with and without the cache, by session length (calculation): 3 turns",93.94,"usd","$93.94","\u0001","\u0001","\u0001","no cache",true,"\u0001","\u0001"],["prompt-cache-break-even","cache-break-even-cost-curve","Claude Opus 5.5 (no cache)","5 turns","cache-break-even-cost-curve/5 turns","Cost of a reused prefix with and without the cache, by session length (calculation): 5 turns",156.56,"usd","$156.56","\u0001","\u0001","\u0001","no cache",true,"\u0001","\u0001"],["prompt-cache-break-even","cache-break-even-cost-curve","Claude Opus 5.5 (no cache)","10 turns","cache-break-even-cost-curve/10 turns","Cost of a reused prefix with and without the cache, by session length (calculation): 10 turns",313.12,"usd","$313.12","\u0001","\u0001","\u0001","no cache",true,"\u0001","\u0001"],["prompt-cache-break-even","cache-break-even-cost-curve","Claude Opus 5.5 (no cache)","20 turns","cache-break-even-cost-curve/20 turns","Cost of a reused prefix with and without the cache, by session length (calculation): 20 turns",626.24,"usd","$626.24","\u0001","\u0001","\u0001","no cache",true,"\u0001","\u0001"],["prompt-cache-break-even","cache-break-even-cost-curve","Claude Opus 5.5 (1-hour cache write, read $0.2 per M)","1 turn","cache-break-even-cost-curve/1 turn","Cost of a reused prefix with and without the cache, by session length (calculation): 1 turn",62.62,"usd","$62.62","\u0001","\u0001","\u0001","1-hour cache write · read $0.2 per M",true,"\u0001","\u0001"],["prompt-cache-break-even","cache-break-even-cost-curve","Claude Opus 5.5 (1-hour cache write, read $0.2 per M)","2 turns","cache-break-even-cost-curve/2 turns","Cost of a reused prefix with and without the cache, by session length (calculation): 2 turns",64.19,"usd","$64.19","\u0001","\u0001","\u0001","1-hour cache write · read $0.2 per M",true,"\u0001","\u0001"],["prompt-cache-break-even","cache-break-even-cost-curve","Claude Opus 5.5 (1-hour cache write, read $0.2 per M)","3 turns","cache-break-even-cost-curve/3 turns","Cost of a reused prefix with and without the cache, by session length (calculation): 3 turns",65.76,"usd","$65.76","\u0001","\u0001","\u0001","1-hour cache write · read $0.2 per M",true,"\u0001","\u0001"],["prompt-cache-break-even","cache-break-even-cost-curve","Claude Opus 5.5 (1-hour cache write, read $0.2 per M)","5 turns","cache-break-even-cost-curve/5 turns","Cost of a reused prefix with and without the cache, by session length (calculation): 5 turns",68.89,"usd","$68.89","\u0001","\u0001","\u0001","1-hour cache write · read $0.2 per M",true,"\u0001","\u0001"],["prompt-cache-break-even","cache-break-even-cost-curve","Claude Opus 5.5 (1-hour cache write, read $0.2 per M)","10 turns","cache-break-even-cost-curve/10 turns","Cost of a reused prefix with and without the cache, by session length (calculation): 10 turns",76.71,"usd","$76.71","\u0001","\u0001","\u0001","1-hour cache write · read $0.2 per M",true,"\u0001","\u0001"],["prompt-cache-break-even","cache-break-even-cost-curve","Claude Opus 5.5 (1-hour cache write, read $0.2 per M)","20 turns","cache-break-even-cost-curve/20 turns","Cost of a reused prefix with and without the cache, by session length (calculation): 20 turns",92.37,"usd","$92.37","\u0001","\u0001","\u0001","1-hour cache write · read $0.2 per M",true,"\u0001","\u0001"],["prompt-cache-break-even","cache-break-even-cost-curve","Claude Opus 5.5 (1-hour cache write, read $0.4 per M)","1 turn","cache-break-even-cost-curve/1 turn","Cost of a reused prefix with and without the cache, by session length (calculation): 1 turn",62.62,"usd","$62.62","\u0001","\u0001","\u0001","1-hour cache write · read $0.4 per M",true,"\u0001","\u0001"],["prompt-cache-break-even","cache-break-even-cost-curve","Claude Opus 5.5 (1-hour cache write, read $0.4 per M)","2 turns","cache-break-even-cost-curve/2 turns","Cost of a reused prefix with and without the cache, by session length (calculation): 2 turns",65.76,"usd","$65.76","\u0001","\u0001","\u0001","1-hour cache write · read $0.4 per M",true,"\u0001","\u0001"],["prompt-cache-break-even","cache-break-even-cost-curve","Claude Opus 5.5 (1-hour cache write, read $0.4 per M)","3 turns","cache-break-even-cost-curve/3 turns","Cost of a reused prefix with and without the cache, by session length (calculation): 3 turns",68.89,"usd","$68.89","\u0001","\u0001","\u0001","1-hour cache write · read $0.4 per M",true,"\u0001","\u0001"],["prompt-cache-break-even","cache-break-even-cost-curve","Claude Opus 5.5 (1-hour cache write, read $0.4 per M)","5 turns","cache-break-even-cost-curve/5 turns","Cost of a reused prefix with and without the cache, by session length (calculation): 5 turns",75.15,"usd","$75.15","\u0001","\u0001","\u0001","1-hour cache write · read $0.4 per M",true,"\u0001","\u0001"],["prompt-cache-break-even","cache-break-even-cost-curve","Claude Opus 5.5 (1-hour cache write, read $0.4 per M)","10 turns","cache-break-even-cost-curve/10 turns","Cost of a reused prefix with and without the cache, by session length (calculation): 10 turns",90.8,"usd","$90.80","\u0001","\u0001","\u0001","1-hour cache write · read $0.4 per M",true,"\u0001","\u0001"],["prompt-cache-break-even","cache-break-even-cost-curve","Claude Opus 5.5 (1-hour cache write, read $0.4 per M)","20 turns","cache-break-even-cost-curve/20 turns","Cost of a reused prefix with and without the cache, by session length (calculation): 20 turns",122.12,"usd","$122.12","\u0001","\u0001","\u0001","1-hour cache write · read $0.4 per M",true,"\u0001","\u0001"],["prompt-cache-break-even","cache-break-even-session-split","No cache (the same either way)","Claude Opus 5.5 (cache read $0.2 per M)","cache-break-even-session-split/No cache (the same either way)","One 10-turn session or ten 1-turn sessions: cost with and without the cache (calculation) (No cache (the same either way))",313.12,"usd","$313.12","\u0001","\u0001","\u0001","cache read $0.2 per M",true,"\u0001","\u0001"],["prompt-cache-break-even","cache-break-even-session-split","No cache (the same either way)","Claude Opus 5.5 (cache read $0.4 per M)","cache-break-even-session-split/No cache (the same either way)","One 10-turn session or ten 1-turn sessions: cost with and without the cache (calculation) (No cache (the same either way))",313.12,"usd","$313.12","\u0001","\u0001","\u0001","cache read $0.4 per M",true,"\u0001","\u0001"],["prompt-cache-break-even","cache-break-even-session-split","One 10-turn session, 1-hour cache","Claude Opus 5.5 (cache read $0.2 per M)","cache-break-even-session-split/One 10-turn session, 1-hour cache","One 10-turn session or ten 1-turn sessions: cost with and without the cache (calculation) (One 10-turn session, 1-hour cache)",76.71,"usd","$76.71","\u0001","\u0001","\u0001","cache read $0.2 per M",true,"\u0001","\u0001"],["prompt-cache-break-even","cache-break-even-session-split","One 10-turn session, 1-hour cache","Claude Opus 5.5 (cache read $0.4 per M)","cache-break-even-session-split/One 10-turn session, 1-hour cache","One 10-turn session or ten 1-turn sessions: cost with and without the cache (calculation) (One 10-turn session, 1-hour cache)",90.8,"usd","$90.80","\u0001","\u0001","\u0001","cache read $0.4 per M",true,"\u0001","\u0001"],["prompt-cache-break-even","cache-break-even-session-split","Ten 1-turn sessions, 1-hour cache (assumed no reuse of the new prefix)","Claude Opus 5.5 (cache read $0.2 per M)","cache-break-even-session-split/Ten 1-turn sessions, 1-hour cache (assumed no reuse of the new prefix)","One 10-turn session or ten 1-turn sessions: cost with and without the cache (calculation) (Ten 1-turn sessions, 1-hour cache (assumed no reuse of the new prefix))",626.24,"usd","$626.24","\u0001","\u0001","\u0001","cache read $0.2 per M",true,"\u0001","\u0001"],["prompt-cache-break-even","cache-break-even-session-split","Ten 1-turn sessions, 1-hour cache (assumed no reuse of the new prefix)","Claude Opus 5.5 (cache read $0.4 per M)","cache-break-even-session-split/Ten 1-turn sessions, 1-hour cache (assumed no reuse of the new prefix)","One 10-turn session or ten 1-turn sessions: cost with and without the cache (calculation) (Ten 1-turn sessions, 1-hour cache (assumed no reuse of the new prefix))",626.24,"usd","$626.24","\u0001","\u0001","\u0001","cache read $0.4 per M",true,"\u0001","\u0001"],["thinking-token-bill","thinking-bill-share","Median call","Claude Opus 5.5 · Claude Code","thinking-bill-share","Reasoning share of output tokens per call on hard tasks (calculation)",54.79,"percent","54.8%",24,"\u0001","minmax","Claude Code",true,[29.92,95.6],"none"],["thinking-token-bill","thinking-bill-share","Median call","Claude Opus 5.5 (high) · Claude Code","thinking-bill-share","Reasoning share of output tokens per call on hard tasks (calculation)",54.43,"percent","54.4%",24,"\u0001","minmax","Claude Code · effort high",true,[36.14,96.23],"none"],["thinking-token-bill","thinking-bill-cost-per-call","Reasoning (output tokens)","Claude Opus 5.5 (high) · Claude Code","thinking-bill-cost-per-call/Reasoning (output tokens)","List-price cost per call: reasoning, remaining output and input (calculation) (Reasoning (output tokens))",0.017969,"usd","$0.018",24,"\u0001","\u0001","Claude Code · effort high",true,"\u0001","none"],["thinking-token-bill","thinking-bill-cost-per-call","Reasoning (output tokens)","Claude Opus 5.5 · Claude Code","thinking-bill-cost-per-call/Reasoning (output tokens)","List-price cost per call: reasoning, remaining output and input (calculation) (Reasoning (output tokens))",0.012528,"usd","$0.013",24,"\u0001","\u0001","Claude Code",true,"\u0001","none"],["thinking-token-bill","thinking-bill-cost-per-call","Remaining output (visible-answer estimate)","Claude Opus 5.5 (high) · Claude Code","thinking-bill-cost-per-call/Remaining output (visible-answer estimate)","List-price cost per call: reasoning, remaining output and input (calculation) (Remaining output (visible-answer estimate))",0.00799,"usd","$0.0080",24,"\u0001","\u0001","Claude Code · effort high",true,"\u0001","none"],["thinking-token-bill","thinking-bill-cost-per-call","Remaining output (visible-answer estimate)","Claude Opus 5.5 · Claude Code","thinking-bill-cost-per-call/Remaining output (visible-answer estimate)","List-price cost per call: reasoning, remaining output and input (calculation) (Remaining output (visible-answer estimate))",0.008003,"usd","$0.0080",24,"\u0001","\u0001","Claude Code",true,"\u0001","none"],["thinking-token-bill","thinking-bill-cost-per-call","Input (prompt, cache priced)","Claude Opus 5.5 (high) · Claude Code","thinking-bill-cost-per-call/Input (prompt, cache priced)","List-price cost per call: reasoning, remaining output and input (calculation) (Input (prompt, cache priced))",0.007407,"usd","$0.0074",24,"\u0001","\u0001","Claude Code · effort high",true,"\u0001","none"],["thinking-token-bill","thinking-bill-cost-per-call","Input (prompt, cache priced)","Claude Opus 5.5 · Claude Code","thinking-bill-cost-per-call/Input (prompt, cache priced)","List-price cost per call: reasoning, remaining output and input (calculation) (Input (prompt, cache priced))",0.00771,"usd","$0.0077",24,"\u0001","\u0001","Claude Code",true,"\u0001","none"],["thinking-token-bill","thinking-bill-by-effort","Reasoning cost per strict pass","Claude Opus 5.5 (low) · Claude Code","thinking-bill-by-effort/Reasoning cost per strict pass","Reasoning cost per strict pass by effort, with the total (calculation) (Reasoning cost per strict pass)",0.005031,"usd","$0.0050",16,"\u0001","\u0001","Claude Code · effort low",true,"\u0001","none"],["thinking-token-bill","thinking-bill-by-effort","Reasoning cost per strict pass","Claude Opus 5.5 (medium) · Claude Code","thinking-bill-by-effort/Reasoning cost per strict pass","Reasoning cost per strict pass by effort, with the total (calculation) (Reasoning cost per strict pass)",0.01344,"usd","$0.013",16,"\u0001","\u0001","Claude Code · effort medium",true,"\u0001","none"],["thinking-token-bill","thinking-bill-by-effort","Reasoning cost per strict pass","Claude Opus 5.5 (high) · Claude Code","thinking-bill-by-effort/Reasoning cost per strict pass","Reasoning cost per strict pass by effort, with the total (calculation) (Reasoning cost per strict pass)",0.018034,"usd","$0.018",16,"\u0001","\u0001","Claude Code · effort high",true,"\u0001","none"],["thinking-token-bill","thinking-bill-by-effort","Reasoning cost per strict pass","Claude Opus 5.5 · Claude Code","thinking-bill-by-effort/Reasoning cost per strict pass","Reasoning cost per strict pass by effort, with the total (calculation) (Reasoning cost per strict pass)",0.013104,"usd","$0.013",16,"\u0001","\u0001","Claude Code",true,"\u0001","none"],["thinking-token-bill","thinking-bill-by-effort","Total cost per strict pass","Claude Opus 5.5 (low) · Claude Code","thinking-bill-by-effort/Total cost per strict pass","Reasoning cost per strict pass by effort, with the total (calculation) (Total cost per strict pass)",0.021152,"usd","$0.021",16,"\u0001","\u0001","Claude Code · effort low",true,"\u0001","none"],["thinking-token-bill","thinking-bill-by-effort","Total cost per strict pass","Claude Opus 5.5 (medium) · Claude Code","thinking-bill-by-effort/Total cost per strict pass","Reasoning cost per strict pass by effort, with the total (calculation) (Total cost per strict pass)",0.029475,"usd","$0.029",16,"\u0001","\u0001","Claude Code · effort medium",true,"\u0001","none"],["thinking-token-bill","thinking-bill-by-effort","Total cost per strict pass","Claude Opus 5.5 (high) · Claude Code","thinking-bill-by-effort/Total cost per strict pass","Reasoning cost per strict pass by effort, with the total (calculation) (Total cost per strict pass)",0.033677,"usd","$0.034",16,"\u0001","\u0001","Claude Code · effort high",true,"\u0001","none"],["thinking-token-bill","thinking-bill-by-effort","Total cost per strict pass","Claude Opus 5.5 · Claude Code","thinking-bill-by-effort/Total cost per strict pass","Reasoning cost per strict pass by effort, with the total (calculation) (Total cost per strict pass)",0.028925,"usd","$0.029",16,"\u0001","\u0001","Claude Code",true,"\u0001","none"],["thinking-token-bill","thinking-bill-short-vs-hard","Eight hard tasks","Claude Opus 5.5 · Claude Code","thinking-bill-short-vs-hard/Eight hard tasks","Reasoning share on short tasks vs hard tasks (calculation) (Eight hard tasks)",54.79,"percent","54.8%",24,"\u0001","minmax","Claude Code",true,[29.92,95.6],"none"],["thinking-token-bill","thinking-bill-short-vs-hard","Eight hard tasks","Claude Opus 5.5 (high) · Claude Code","thinking-bill-short-vs-hard/Eight hard tasks","Reasoning share on short tasks vs hard tasks (calculation) (Eight hard tasks)",54.43,"percent","54.4%",24,"\u0001","minmax","Claude Code · effort high",true,[36.14,96.23],"none"],["thinking-token-bill","thinking-bill-short-vs-hard","Five short tasks","Claude Opus 5.5 · Claude Code","thinking-bill-short-vs-hard/Five short tasks","Reasoning share on short tasks vs hard tasks (calculation) (Five short tasks)",0,"percent","0%",15,"\u0001","minmax","Claude Code",true,[0,93.33],"none"],["thinking-token-bill","thinking-bill-short-vs-hard","Five short tasks","Claude Opus 5.5 (high) · Claude Code","thinking-bill-short-vs-hard/Five short tasks","Reasoning share on short tasks vs hard tasks (calculation) (Five short tasks)",43.59,"percent","43.6%",15,"\u0001","minmax","Claude Code · effort high",true,[0,93.33],"none"],["llm-speed-anatomy","speed-anatomy-first-text","Time to first text","Claude Opus 5.5 · Claude Code","speed-anatomy-first-text","Time to first text: a 250-line answer, six models",1.97,"seconds","1.97 s",4,"\u0001","minmax","Claude Code","\u0001",[1.7,2.35],"\u0001"],["llm-speed-anatomy","speed-anatomy-output-speed","Visible tokens per second","Claude Opus 5.5 · Claude Code","speed-anatomy-output-speed","Output speed after the first text: visible tokens per second (calculation)",155.5,"tokens","156",4,"\u0001","minmax","Claude Code",true,[154.6,156.4],"\u0001"],["llm-speed-anatomy","speed-anatomy-chars-per-second","Characters per second","Claude Opus 5.5 · Claude Code","speed-anatomy-chars-per-second","Output speed in characters per second after the first text (calculation)",347,"count","347",4,"\u0001","minmax","Claude Code",true,[345,349],"\u0001"],["llm-speed-anatomy","speed-anatomy-prompt-size","Claude Opus 5.5 · Claude Code","1k","speed-anatomy-prompt-size/1k","Time to first text as the prompt grows: 1k",1.51,"seconds","1.51 s",3,"\u0001","minmax","Claude Code",true,[1.46,2.01],"\u0001"],["llm-speed-anatomy","speed-anatomy-prompt-size","Claude Opus 5.5 · Claude Code","16k","speed-anatomy-prompt-size/16k","Time to first text as the prompt grows: 16k",1.74,"seconds","1.74 s",3,"\u0001","minmax","Claude Code",true,[1.7,2.97],"\u0001"],["llm-speed-anatomy","speed-anatomy-prompt-size","Claude Opus 5.5 · Claude Code","64k","speed-anatomy-prompt-size/64k","Time to first text as the prompt grows: 64k",1.79,"seconds","1.79 s",3,"\u0001","minmax","Claude Code",true,[1.72,3.72],"\u0001"],["llm-speed-anatomy","speed-anatomy-total-by-size","1k prompt","Claude Opus 5.5 · Claude Code","speed-anatomy-total-by-size/1k prompt","Total time per call by prompt size (1k prompt)",1.83,"seconds","1.83 s",3,"\u0001","minmax","Claude Code","\u0001",[1.82,2.41],"\u0001"],["llm-speed-anatomy","speed-anatomy-total-by-size","16k prompt","Claude Opus 5.5 · Claude Code","speed-anatomy-total-by-size/16k prompt","Total time per call by prompt size (16k prompt)",2.36,"seconds","2.36 s",3,"\u0001","minmax","Claude Code","\u0001",[2.11,3.4],"\u0001"],["llm-speed-anatomy","speed-anatomy-total-by-size","64k prompt","Claude Opus 5.5 · Claude Code","speed-anatomy-total-by-size/64k prompt","Total time per call by prompt size (64k prompt)",2.35,"seconds","2.35 s",3,"\u0001","minmax","Claude Code","\u0001",[2.26,4.29],"\u0001"],["llm-speed-anatomy","speed-anatomy-lookup-correct","Exact answer","Claude Opus 5.5 · Claude Code","speed-anatomy-lookup-correct","Exact lookup answers at the 1k, 16k and 64k prompt-size targets",0.5556,"rate","56% (5/9)",9,[0.2667,0.8112],"ci95","Claude Code","\u0001","\u0001","higher"],["harder-tasks-head-to-head","harder-h2h-pass-rate","Strict pass","Claude Opus 5.5 · Claude Code","harder-h2h-pass-rate/Strict pass","Pass rate on 4 harder tasks (Strict pass)",0.4167,"rate","42% (5/12)",12,[0.1933,0.6805],"ci95","Claude Code","\u0001","\u0001","higher"],["harder-tasks-head-to-head","harder-h2h-pass-rate","Lenient (format misses counted)","Claude Opus 5.5 · Claude Code","harder-h2h-pass-rate/Lenient (format misses counted)","Pass rate on 4 harder tasks (Lenient (format misses counted))",0.5,"rate","50% (6/12)",12,[0.2538,0.7462],"ci95","Claude Code","\u0001","\u0001","higher"],["harder-tasks-head-to-head","harder-h2h-tool-attempts","Tool attempt","Claude Opus 5.5 · Claude Code","harder-h2h-tool-attempts","Calls that tried a tool although tools were off",0.4167,"rate","42% (5/12)",12,[0.1933,0.6805],"ci95","Claude Code","\u0001","\u0001","none"],["harder-tasks-head-to-head","harder-h2h-pass-by-task","Claude Opus 5.5 · Claude Code","10x10 nonogram","harder-h2h-pass-by-task/10x10 nonogram","Strict pass rate by task: 10x10 nonogram",1,"rate","100% (3/3)",3,[0.4385,1],"ci95","Claude Code","\u0001","\u0001","higher"],["harder-tasks-head-to-head","harder-h2h-pass-by-task","Claude Opus 5.5 · Claude Code","Sudoku, 22 givens","harder-h2h-pass-by-task/Sudoku, 22 givens","Strict pass rate by task: Sudoku, 22 givens",0,"rate","0% (0/3)",3,[0,0.5615],"ci95","Claude Code","\u0001","\u0001","higher"],["harder-tasks-head-to-head","harder-h2h-pass-by-task","Claude Opus 5.5 · Claude Code","6x6 Skyscrapers","harder-h2h-pass-by-task/6x6 Skyscrapers","Strict pass rate by task: 6x6 Skyscrapers",0.3333,"rate","33% (1/3)",3,[0.0615,0.7923],"ci95","Claude Code","\u0001","\u0001","higher"],["harder-tasks-head-to-head","harder-h2h-pass-by-task","Claude Opus 5.5 · Claude Code","Seeded shuffle output","harder-h2h-pass-by-task/Seeded shuffle output","Strict pass rate by task: Seeded shuffle output",0.3333,"rate","33% (1/3)",3,[0.0615,0.7923],"ci95","Claude Code","\u0001","\u0001","higher"],["harder-tasks-head-to-head","harder-h2h-total-latency","Total time per call","Claude Opus 5.5 · Claude Code","harder-h2h-total-latency","Total time per call on harder tasks",80.34,"seconds","80.3 s",9,"\u0001","minmax","Claude Code","\u0001",[3.82,279.5],"\u0001"],["harder-tasks-head-to-head","harder-h2h-output-tokens","Output tokens","Claude Opus 5.5 · Claude Code","harder-h2h-output-tokens/Output tokens","Output tokens per call on harder tasks (Output tokens)",8420,"tokens","8,420",9,"\u0001","minmax","Claude Code","\u0001",[279,40044],"none"],["harder-tasks-head-to-head","harder-h2h-cost-per-pass","Cost per strict pass","Claude Opus 5.5 · Claude Code","harder-h2h-cost-per-pass","List-price cost per strict pass on harder tasks (calculation)",0.59333,"usd","$0.59",12,"\u0001","\u0001","Claude Code",true,"\u0001","\u0001"]]}],["claude-haiku-4-5","Claude Haiku 4.5","Anthropic","model","Anthropic’s small, low-price Claude model, measured through Claude Code, as an LLM router and in the public SWE-bench panel.",["Claude Haiku 4.5","Haiku 4.5","Claude 4.5 Haiku"],{"$k":["studySlug","chartId","series","point","metric","label","value","unit","display","n","ci","spanKind","context","range","calculation","polarity","statId"],"$r":[["swe-bench-verified","swebench-same-instance-leaderboard","Resolved rate","Claude 4.5 Haiku (high)","swebench-same-instance-leaderboard","Resolved rate on the same 33 SWE-bench Verified instances",0.7576,"rate","76% (25/33)",33,[0.5898,0.8717],"ci95","effort high · public mini-SWE-agent v2 run, same instances","\u0001","\u0001","\u0001","\u0001"],["swe-bench-verified","swebench-model-calls","Mean calls","Claude 4.5 Haiku (high)","swebench-model-calls","Model calls per instance",68.5,"calls","68.5",33,"\u0001","\u0001","effort high · public mini-SWE-agent v2 run, same instances","\u0001","\u0001","\u0001","\u0001"],["model-head-to-head","h2h-pass-rate","Pass rate","Claude Haiku 4.5 · Claude Code","h2h-pass-rate","Pass rate on five validated tasks",1,"rate","100% (15/15)",15,[0.7961,1],"ci95","Claude Code · five short validated tasks","\u0001","\u0001","\u0001","\u0001"],["model-head-to-head","h2h-total-latency","Total time per call","Claude Haiku 4.5 · Claude Code","h2h-total-latency","Total time per call",4.43,"seconds","4.43 s",15,"\u0001","minmax","Claude Code · five short validated tasks",[3.16,23.57],"\u0001","\u0001","\u0001"],["model-head-to-head","h2h-first-useful-latency","Time to first useful output","Claude Haiku 4.5 · Claude Code","h2h-first-useful-latency","Time to first useful output",3.63,"seconds","3.63 s",15,"\u0001","minmax","Claude Code · five short validated tasks",[2.78,22.27],"\u0001","\u0001","\u0001"],["model-head-to-head","h2h-input-tokens","Cache read","Claude Haiku 4.5 · Claude Code","h2h-input-tokens/Cache read","Input tokens per call: what the CLI sends (Cache read)",0,"tokens","0",15,"\u0001","\u0001","Claude Code · five short validated tasks","\u0001","\u0001","\u0001","\u0001"],["model-head-to-head","h2h-input-tokens","Other input","Claude Haiku 4.5 · Claude Code","h2h-input-tokens/Other input","Input tokens per call: what the CLI sends (Other input)",3790,"tokens","3,790",15,"\u0001","\u0001","Claude Code · five short validated tasks","\u0001","\u0001","\u0001","\u0001"],["model-head-to-head","h2h-output-tokens","Output tokens","Claude Haiku 4.5 · Claude Code","h2h-output-tokens/Output tokens","Output tokens per call (Output tokens)",367,"tokens","367",15,"\u0001","\u0001","Claude Code · five short validated tasks","\u0001","\u0001","\u0001","\u0001"],["model-head-to-head","h2h-list-price-per-call","Cost per call","Claude Haiku 4.5 · Claude Code","h2h-list-price-per-call","List-price cost per call (calculation)",0.00566,"usd","$0.0057",15,"\u0001","minmax","Claude Code · five short validated tasks",[0.00513,0.01804],true,"\u0001","\u0001"],["model-head-to-head","h2h-cost-per-pass","Cost per pass","Claude Haiku 4.5 · Claude Code","h2h-cost-per-pass","List-price cost per passing answer (calculation)",0.00836,"usd","$0.0084",15,"\u0001","\u0001","Claude Code · five short validated tasks","\u0001",true,"\u0001","\u0001"],["hard-model-head-to-head","hard-h2h-pass-rate","Strict pass","Claude Haiku 4.5 · Claude Code","hard-h2h-pass-rate/Strict pass","Pass rate on eight hard tasks (Strict pass)",0.4583,"rate","46% (11/24)",24,[0.2789,0.6493],"ci95","Claude Code · eight hard validated tasks","\u0001","\u0001","\u0001","\u0001"],["hard-model-head-to-head","hard-h2h-pass-rate","Lenient (format misses counted)","Claude Haiku 4.5 · Claude Code","hard-h2h-pass-rate/Lenient (format misses counted)","Pass rate on eight hard tasks (Lenient (format misses counted))",0.6667,"rate","67% (16/24)",24,[0.4671,0.8203],"ci95","Claude Code · eight hard validated tasks","\u0001","\u0001","\u0001","\u0001"],["hard-model-head-to-head","hard-h2h-total-latency","Total time per call on hard tasks (separate batches)","Claude Haiku 4.5 · Claude Code","hard-h2h-total-latency","Total time per call on hard tasks (separate batches)",39.01,"seconds","39.0 s",24,"\u0001","minmax","Claude Code · eight hard validated tasks",[15.27,75.13],"\u0001","\u0001","\u0001"],["hard-model-head-to-head","hard-h2h-first-useful-latency","Time to first useful output on hard tasks","Claude Haiku 4.5 · Claude Code","hard-h2h-first-useful-latency","Time to first useful output on hard tasks",35.54,"seconds","35.5 s",24,"\u0001","minmax","Claude Code · eight hard validated tasks",[12.88,70.31],"\u0001","\u0001","\u0001"],["hard-model-head-to-head","hard-h2h-output-tokens","Output tokens","Claude Haiku 4.5 · Claude Code","hard-h2h-output-tokens/Output tokens","Output tokens per call on hard tasks (Output tokens)",5064,"tokens","5,064",24,"\u0001","\u0001","Claude Code · eight hard validated tasks","\u0001","\u0001","\u0001","\u0001"],["hard-model-head-to-head","hard-h2h-cost-per-pass","Cost per strict pass","Claude Haiku 4.5 · Claude Code","hard-h2h-cost-per-pass","List-price cost per strict pass on hard tasks (calculation)",0.0672,"usd","$0.067",24,"\u0001","\u0001","Claude Code · eight hard validated tasks","\u0001",true,"\u0001","\u0001"],["caching-consistency","consistency-pass-rate","Exact number","Claude Haiku 4.5 · Claude Code","consistency-pass-rate/Exact number","Same prompt, 10 times: strict pass rate (Exact number)",0,"rate","0% (0/10)",10,[0,0.2775],"ci95","Claude Code · same prompt repeated 10 times","\u0001","\u0001","\u0001","\u0001"],["caching-consistency","consistency-pass-rate","JSON object","Claude Haiku 4.5 · Claude Code","consistency-pass-rate/JSON object","Same prompt, 10 times: strict pass rate (JSON object)",0.1,"rate","10% (1/10)",10,[0.0179,0.4042],"ci95","Claude Code · same prompt repeated 10 times","\u0001","\u0001","\u0001","\u0001"],["caching-consistency","consistency-pass-rate","Code fix","Claude Haiku 4.5 · Claude Code","consistency-pass-rate/Code fix","Same prompt, 10 times: strict pass rate (Code fix)",1,"rate","100% (10/10)",10,[0.7225,1],"ci95","Claude Code · same prompt repeated 10 times","\u0001","\u0001","\u0001","\u0001"],["caching-consistency","consistency-distinct-answers","Exact number","Claude Haiku 4.5 · Claude Code","consistency-distinct-answers/Exact number","Same prompt, 10 times: how many different answers (Exact number)",1,"count","1",10,"\u0001","\u0001","Claude Code · same prompt repeated 10 times","\u0001","\u0001","\u0001","\u0001"],["caching-consistency","consistency-distinct-answers","JSON object","Claude Haiku 4.5 · Claude Code","consistency-distinct-answers/JSON object","Same prompt, 10 times: how many different answers (JSON object)",1,"count","1",10,"\u0001","\u0001","Claude Code · same prompt repeated 10 times","\u0001","\u0001","\u0001","\u0001"],["caching-consistency","consistency-distinct-answers","Code fix","Claude Haiku 4.5 · Claude Code","consistency-distinct-answers/Code fix","Same prompt, 10 times: how many different answers (Code fix)",6,"count","6",10,"\u0001","\u0001","Claude Code · same prompt repeated 10 times","\u0001","\u0001","\u0001","\u0001"],["caching-consistency","consistency-latency-spread","Exact number","Claude Haiku 4.5 · Claude Code","consistency-latency-spread/Exact number","Same prompt, 10 times: time per call (Exact number)",5.06,"seconds","5.06 s",10,"\u0001","minmax","Claude Code · same prompt repeated 10 times",[4.42,6.2],"\u0001","\u0001","\u0001"],["caching-consistency","consistency-latency-spread","JSON object","Claude Haiku 4.5 · Claude Code","consistency-latency-spread/JSON object","Same prompt, 10 times: time per call (JSON object)",7.03,"seconds","7.03 s",10,"\u0001","minmax","Claude Code · same prompt repeated 10 times",[5.28,12.27],"\u0001","\u0001","\u0001"],["caching-consistency","consistency-latency-spread","Code fix","Claude Haiku 4.5 · Claude Code","consistency-latency-spread/Code fix","Same prompt, 10 times: time per call (Code fix)",5.95,"seconds","5.95 s",10,"\u0001","minmax","Claude Code · same prompt repeated 10 times",[4.89,7.33],"\u0001","\u0001","\u0001"],["agent-memory","memory-full-pass","Claude Haiku 4.5","No memory","memory-full-pass/No memory","Full pass rate by kind of memory: No memory",0.2,"rate","20% (2/10)",10,[0.0567,0.5098],"ci95","","\u0001","\u0001","\u0001","\u0001"],["agent-memory","memory-full-pass","Claude Haiku 4.5","/init CLAUDE.md","memory-full-pass//init CLAUDE.md","Full pass rate by kind of memory: /init CLAUDE.md",0.2,"rate","20% (2/10)",10,[0.0567,0.5098],"ci95","","\u0001","\u0001","\u0001","\u0001"],["agent-memory","memory-full-pass","Claude Haiku 4.5","Curated, 11 lines","memory-full-pass/Curated, 11 lines","Full pass rate by kind of memory: Curated, 11 lines",0.7,"rate","70% (7/10)",10,[0.3968,0.8922],"ci95","","\u0001","\u0001","\u0001","\u0001"],["agent-memory","memory-full-pass","Claude Haiku 4.5","Raw notes, 60 lines","memory-full-pass/Raw notes, 60 lines","Full pass rate by kind of memory: Raw notes, 60 lines",0.6,"rate","60% (6/10)",10,[0.3127,0.8318],"ci95","","\u0001","\u0001","\u0001","\u0001"],["agent-memory","memory-full-pass","Claude Haiku 4.5","Dreamed notes","memory-full-pass/Dreamed notes","Full pass rate by kind of memory: Dreamed notes",0.7,"rate","70% (7/10)",10,[0.3968,0.8922],"ci95","","\u0001","\u0001","\u0001","\u0001"],["agent-memory","memory-full-pass","Claude Haiku 4.5","Handbook, 210 lines","memory-full-pass/Handbook, 210 lines","Full pass rate by kind of memory: Handbook, 210 lines",0.3,"rate","30% (3/10)",10,[0.1078,0.6032],"ci95","","\u0001","\u0001","\u0001","\u0001"],["agent-memory","memory-full-pass","Claude Haiku 4.5","Stop hook only","memory-full-pass/Stop hook only","Full pass rate by kind of memory: Stop hook only",0.8,"rate","80% (8/10)",10,[0.4902,0.9433],"ci95","","\u0001","\u0001","\u0001","\u0001"],["agent-memory","memory-full-pass","Claude Haiku 4.5","Curated + hook","memory-full-pass/Curated + hook","Full pass rate by kind of memory: Curated + hook",0.9,"rate","90% (9/10)",10,[0.5958,0.9821],"ci95","","\u0001","\u0001","\u0001","\u0001"],["agent-memory","memory-team-knowledge-by-model","Claude Haiku 4.5","No memory","memory-team-knowledge-by-model/No memory","Team knowledge followed, Sonnet vs Haiku: No memory",0,"rate","0% (0/10)",10,[0,0.2775],"ci95","","\u0001","\u0001","\u0001","\u0001"],["agent-memory","memory-team-knowledge-by-model","Claude Haiku 4.5","/init CLAUDE.md","memory-team-knowledge-by-model//init CLAUDE.md","Team knowledge followed, Sonnet vs Haiku: /init CLAUDE.md",0.1,"rate","10% (1/10)",10,[0.0179,0.4042],"ci95","","\u0001","\u0001","\u0001","\u0001"],["agent-memory","memory-team-knowledge-by-model","Claude Haiku 4.5","Curated, 11 lines","memory-team-knowledge-by-model/Curated, 11 lines","Team knowledge followed, Sonnet vs Haiku: Curated, 11 lines",0.8,"rate","80% (8/10)",10,[0.4902,0.9433],"ci95","","\u0001","\u0001","\u0001","\u0001"],["agent-memory","memory-team-knowledge-by-model","Claude Haiku 4.5","Raw notes, 60 lines","memory-team-knowledge-by-model/Raw notes, 60 lines","Team knowledge followed, Sonnet vs Haiku: Raw notes, 60 lines",0.6,"rate","60% (6/10)",10,[0.3127,0.8318],"ci95","","\u0001","\u0001","\u0001","\u0001"],["agent-memory","memory-team-knowledge-by-model","Claude Haiku 4.5","Dreamed notes","memory-team-knowledge-by-model/Dreamed notes","Team knowledge followed, Sonnet vs Haiku: Dreamed notes",0.8,"rate","80% (8/10)",10,[0.4902,0.9433],"ci95","","\u0001","\u0001","\u0001","\u0001"],["agent-memory","memory-team-knowledge-by-model","Claude Haiku 4.5","Handbook, 210 lines","memory-team-knowledge-by-model/Handbook, 210 lines","Team knowledge followed, Sonnet vs Haiku: Handbook, 210 lines",0.3,"rate","30% (3/10)",10,[0.1078,0.6032],"ci95","","\u0001","\u0001","\u0001","\u0001"],["agent-memory","memory-team-knowledge-by-model","Claude Haiku 4.5","Stop hook only","memory-team-knowledge-by-model/Stop hook only","Team knowledge followed, Sonnet vs Haiku: Stop hook only",0.8,"rate","80% (8/10)",10,[0.4902,0.9433],"ci95","","\u0001","\u0001","\u0001","\u0001"],["agent-memory","memory-team-knowledge-by-model","Claude Haiku 4.5","Curated + hook","memory-team-knowledge-by-model/Curated + hook","Team knowledge followed, Sonnet vs Haiku: Curated + hook",1,"rate","100% (10/10)",10,[0.7225,1],"ci95","","\u0001","\u0001","\u0001","\u0001"],["agent-memory","memory-broken-test-command","Claude Haiku 4.5","No memory","memory-broken-test-command/No memory","A stale README command: who still ran it?: No memory",1,"rate","100% (10/10)",10,[0.7225,1],"ci95","","\u0001","\u0001","lower","\u0001"],["agent-memory","memory-broken-test-command","Claude Haiku 4.5","/init CLAUDE.md","memory-broken-test-command//init CLAUDE.md","A stale README command: who still ran it?: /init CLAUDE.md",1,"rate","100% (10/10)",10,[0.7225,1],"ci95","","\u0001","\u0001","lower","\u0001"],["agent-memory","memory-broken-test-command","Claude Haiku 4.5","Curated, 11 lines","memory-broken-test-command/Curated, 11 lines","A stale README command: who still ran it?: Curated, 11 lines",0,"rate","0% (0/10)",10,[0,0.2775],"ci95","","\u0001","\u0001","lower","\u0001"],["agent-memory","memory-broken-test-command","Claude Haiku 4.5","Raw notes, 60 lines","memory-broken-test-command/Raw notes, 60 lines","A stale README command: who still ran it?: Raw notes, 60 lines",1,"rate","100% (10/10)",10,[0.7225,1],"ci95","","\u0001","\u0001","lower","\u0001"],["agent-memory","memory-broken-test-command","Claude Haiku 4.5","Dreamed notes","memory-broken-test-command/Dreamed notes","A stale README command: who still ran it?: Dreamed notes",0.1,"rate","10% (1/10)",10,[0.0179,0.4042],"ci95","","\u0001","\u0001","lower","\u0001"],["agent-memory","memory-broken-test-command","Claude Haiku 4.5","Handbook, 210 lines","memory-broken-test-command/Handbook, 210 lines","A stale README command: who still ran it?: Handbook, 210 lines",0,"rate","0% (0/10)",10,[0,0.2775],"ci95","","\u0001","\u0001","lower","\u0001"],["agent-memory","memory-broken-test-command","Claude Haiku 4.5","Stop hook only","memory-broken-test-command/Stop hook only","A stale README command: who still ran it?: Stop hook only",1,"rate","100% (10/10)",10,[0.7225,1],"ci95","","\u0001","\u0001","lower","\u0001"],["agent-memory","memory-broken-test-command","Claude Haiku 4.5","Curated + hook","memory-broken-test-command/Curated + hook","A stale README command: who still ran it?: Curated + hook",0,"rate","0% (0/10)",10,[0,0.2775],"ci95","","\u0001","\u0001","lower","\u0001"],["agent-memory","memory-cost-per-full-pass","Claude Haiku 4.5","No memory","memory-cost-per-full-pass/No memory","List-price cost per fully correct result (calculation): No memory",0.3786,"usd","$0.38",2,"\u0001","\u0001","","\u0001",true,"\u0001","\u0001"],["agent-memory","memory-cost-per-full-pass","Claude Haiku 4.5","/init CLAUDE.md","memory-cost-per-full-pass//init CLAUDE.md","List-price cost per fully correct result (calculation): /init CLAUDE.md",0.4317,"usd","$0.43",2,"\u0001","\u0001","","\u0001",true,"\u0001","\u0001"],["agent-memory","memory-cost-per-full-pass","Claude Haiku 4.5","Curated, 11 lines","memory-cost-per-full-pass/Curated, 11 lines","List-price cost per fully correct result (calculation): Curated, 11 lines",0.1095,"usd","$0.11",7,"\u0001","\u0001","","\u0001",true,"\u0001","\u0001"],["agent-memory","memory-cost-per-full-pass","Claude Haiku 4.5","Raw notes, 60 lines","memory-cost-per-full-pass/Raw notes, 60 lines","List-price cost per fully correct result (calculation): Raw notes, 60 lines",0.1255,"usd","$0.13",6,"\u0001","\u0001","","\u0001",true,"\u0001","\u0001"],["agent-memory","memory-cost-per-full-pass","Claude Haiku 4.5","Dreamed notes","memory-cost-per-full-pass/Dreamed notes","List-price cost per fully correct result (calculation): Dreamed notes",0.1153,"usd","$0.12",7,"\u0001","\u0001","","\u0001",true,"\u0001","\u0001"],["agent-memory","memory-cost-per-full-pass","Claude Haiku 4.5","Handbook, 210 lines","memory-cost-per-full-pass/Handbook, 210 lines","List-price cost per fully correct result (calculation): Handbook, 210 lines",0.2615,"usd","$0.26",3,"\u0001","\u0001","","\u0001",true,"\u0001","\u0001"],["agent-memory","memory-cost-per-full-pass","Claude Haiku 4.5","Stop hook only","memory-cost-per-full-pass/Stop hook only","List-price cost per fully correct result (calculation): Stop hook only",0.1419,"usd","$0.14",8,"\u0001","\u0001","","\u0001",true,"\u0001","\u0001"],["agent-memory","memory-cost-per-full-pass","Claude Haiku 4.5","Curated + hook","memory-cost-per-full-pass/Curated + hook","List-price cost per fully correct result (calculation): Curated + hook",0.096,"usd","$0.096",9,"\u0001","\u0001","","\u0001",true,"\u0001","\u0001"],["agent-memory","memory-wall-time","Claude Haiku 4.5","No memory","memory-wall-time/No memory","Time per session: No memory",54.2,"seconds","54.2 s",10,"\u0001","\u0001","","\u0001","\u0001","\u0001","\u0001"],["agent-memory","memory-wall-time","Claude Haiku 4.5","/init CLAUDE.md","memory-wall-time//init CLAUDE.md","Time per session: /init CLAUDE.md",52.9,"seconds","52.9 s",10,"\u0001","\u0001","","\u0001","\u0001","\u0001","\u0001"],["agent-memory","memory-wall-time","Claude Haiku 4.5","Curated, 11 lines","memory-wall-time/Curated, 11 lines","Time per session: Curated, 11 lines",51.7,"seconds","51.7 s",10,"\u0001","\u0001","","\u0001","\u0001","\u0001","\u0001"],["agent-memory","memory-wall-time","Claude Haiku 4.5","Raw notes, 60 lines","memory-wall-time/Raw notes, 60 lines","Time per session: Raw notes, 60 lines",51,"seconds","51.0 s",10,"\u0001","\u0001","","\u0001","\u0001","\u0001","\u0001"],["agent-memory","memory-wall-time","Claude Haiku 4.5","Dreamed notes","memory-wall-time/Dreamed notes","Time per session: Dreamed notes",51.8,"seconds","51.8 s",10,"\u0001","\u0001","","\u0001","\u0001","\u0001","\u0001"],["agent-memory","memory-wall-time","Claude Haiku 4.5","Handbook, 210 lines","memory-wall-time/Handbook, 210 lines","Time per session: Handbook, 210 lines",49.9,"seconds","49.9 s",10,"\u0001","\u0001","","\u0001","\u0001","\u0001","\u0001"],["agent-memory","memory-wall-time","Claude Haiku 4.5","Stop hook only","memory-wall-time/Stop hook only","Time per session: Stop hook only",68.5,"seconds","68.5 s",10,"\u0001","\u0001","","\u0001","\u0001","\u0001","\u0001"],["agent-memory","memory-wall-time","Claude Haiku 4.5","Curated + hook","memory-wall-time/Curated + hook","Time per session: Curated + hook",52.8,"seconds","52.8 s",10,"\u0001","\u0001","","\u0001","\u0001","\u0001","\u0001"],["routing-jev-vs-llm","routing-exact-decisions","Exact rate","Claude Haiku 4.5","routing-exact-decisions","Typed routing decisions answered exactly right",0.8902,"rate","89% (73/82)",82,[0.8044,0.9412],"ci95","typed routing decisions · via Claude Code","\u0001","\u0001","\u0001","\u0001"],["routing-jev-vs-llm","routing-key-accuracy","Key accuracy","Claude Haiku 4.5","routing-key-accuracy","Per-question accuracy",0.9433,"rate","94% (183/194)",194,[0.9013,0.968],"ci95","typed routing decisions · via Claude Code","\u0001","\u0001","\u0001","\u0001"],["routing-jev-vs-llm","routing-exact-by-decision","Claude Haiku 4.5","Failure class","routing-exact-by-decision/Failure class","Exact rate by decision type: Failure class",0.9444,"rate","94% (17/18)",18,[0.7424,0.9901],"ci95","typed routing decisions · via Claude Code","\u0001","\u0001","\u0001","\u0001"],["routing-jev-vs-llm","routing-exact-by-decision","Claude Haiku 4.5","Message intent","routing-exact-by-decision/Message intent","Exact rate by decision type: Message intent",1,"rate","100% (20/20)",20,[0.8389,1],"ci95","typed routing decisions · via Claude Code","\u0001","\u0001","\u0001","\u0001"],["routing-jev-vs-llm","routing-exact-by-decision","Claude Haiku 4.5","Is it a rule?","routing-exact-by-decision/Is it a rule?","Exact rate by decision type: Is it a rule?",1,"rate","100% (12/12)",12,[0.7575,1],"ci95","typed routing decisions · via Claude Code","\u0001","\u0001","\u0001","\u0001"],["routing-jev-vs-llm","routing-exact-by-decision","Claude Haiku 4.5","Context shape","routing-exact-by-decision/Context shape","Exact rate by decision type: Context shape",0.75,"rate","75% (24/32)",32,[0.5789,0.8675],"ci95","typed routing decisions · via Claude Code","\u0001","\u0001","\u0001","\u0001"],["routing-jev-vs-llm","routing-cost-per-1000","Cost","Claude Haiku 4.5","routing-cost-per-1000","Cost per 1,000 routing decisions",8.924,"usd","$8.92",82,"\u0001","\u0001","typed routing decisions · via Claude Code","\u0001",true,"\u0001","\u0001"],["routing-jev-vs-llm","routing-decision-latency","Wall time (CLI)","Claude Haiku 4.5","routing-decision-latency/Wall time (CLI)","Time per routing decision (Wall time (CLI))",12674,"ms","12,674 ms",82,"\u0001","p50-p95","typed routing decisions · via Claude Code",[12674,34413],"\u0001","\u0001","\u0001"],["routing-jev-vs-llm","routing-decision-latency","Model time (API)","Claude Haiku 4.5","routing-decision-latency/Model time (API)","Time per routing decision (Model time (API))",10734,"ms","10,734 ms",82,"\u0001","p50-p95","typed routing decisions · via Claude Code",[10734,32072],"\u0001","\u0001","\u0001"],["routing-overhead","router-overhead-decision-latency","Decision time","Claude Haiku 4.5 (thinking on, via Claude Code)","router-overhead-decision-latency","Time to make one routing decision",12543,"ms","12,543 ms",82,"\u0001","p50-p95","thinking on · via Claude Code · routing overhead per decision",[12543,34481],"\u0001","\u0001","\u0001"],["routing-overhead","router-overhead-cli-vs-model-time","Model API time","Claude Haiku 4.5 (thinking on, via Claude Code)","router-overhead-cli-vs-model-time/Model API time","Where an LLM router’s time goes: model vs CLI (Model API time)",10508,"ms","10,508 ms",82,"\u0001","p50-p95","thinking on · via Claude Code · routing overhead per decision",[10508,32132],"\u0001","\u0001","\u0001"],["routing-overhead","router-overhead-cli-vs-model-time","CLI and harness time","Claude Haiku 4.5 (thinking on, via Claude Code)","router-overhead-cli-vs-model-time/CLI and harness time","Where an LLM router’s time goes: model vs CLI (CLI and harness time)",1698,"ms","1,698 ms",82,"\u0001","p50-p95","thinking on · via Claude Code · routing overhead per decision",[1698,2677],"\u0001","\u0001","\u0001"],["routing-overhead","router-overhead-completed","Completed","Claude Haiku 4.5 (thinking on, via Claude Code)","router-overhead-completed","Routing calls that returned a decision",1,"rate","100% (82/82)",82,[0.9552,1],"ci95","thinking on · via Claude Code · routing overhead per decision","\u0001","\u0001","\u0001","\u0001"],["routing-overhead","router-overhead-cost-list-price","Cost per 1,000 decisions (list price)","Claude Haiku 4.5 (thinking on, via Claude Code)","router-overhead-cost-list-price","Cost per 1,000 routing decisions for the model routers (calculation)",8.924,"usd","$8.92",82,"\u0001","\u0001","thinking on · via Claude Code · calculation: Agent’s recorded tokens at this model’s list price · routing overhead per decision","\u0001",true,"\u0001","\u0001"],["routing-overhead","router-overhead-cost-per-1000-tasks","Every model call routed (49.5 per task)","Claude Haiku 4.5 (thinking on, via Claude Code)","router-overhead-cost-per-1000-tasks/Every model call routed (49.5 per task)","Added routing cost per 1,000 tasks (calculation) (Every model call routed (49.5 per task))",441.74,"usd","$441.74","\u0001","\u0001","\u0001","thinking on · via Claude Code · calculation per 1,000 tasks from recorded decision counts","\u0001",true,"\u0001","\u0001"],["routing-overhead","router-overhead-cost-per-1000-tasks","Only System One decisions (7 per task)","Claude Haiku 4.5 (thinking on, via Claude Code)","router-overhead-cost-per-1000-tasks/Only System One decisions (7 per task)","Added routing cost per 1,000 tasks (calculation) (Only System One decisions (7 per task))",62.47,"usd","$62.47","\u0001","\u0001","\u0001","thinking on · via Claude Code · calculation per 1,000 tasks from recorded decision counts","\u0001",true,"\u0001","\u0001"],["routing-overhead","router-overhead-delay-per-task","Every model call routed (49.5 per task)","Claude Haiku 4.5 (thinking on, via Claude Code)","router-overhead-delay-per-task/Every model call routed (49.5 per task)","Added routing delay per task (calculation) (Every model call routed (49.5 per task))",620.8785,"seconds","620.9 s","\u0001","\u0001","\u0001","thinking on · via Claude Code · calculation per task from recorded decision counts, decisions in line","\u0001",true,"\u0001","\u0001"],["routing-overhead","router-overhead-delay-per-task","Only System One decisions (7 per task)","Claude Haiku 4.5 (thinking on, via Claude Code)","router-overhead-delay-per-task/Only System One decisions (7 per task)","Added routing delay per task (calculation) (Only System One decisions (7 per task))",87.801,"seconds","87.8 s","\u0001","\u0001","\u0001","thinking on · via Claude Code · calculation per task from recorded decision counts, decisions in line","\u0001",true,"\u0001","\u0001"],["routing-overhead","cli-startup-tax","First output event","Claude Code · Claude Haiku 4.5","cli-startup-tax/First output event","CLI start-up tax on a one-word answer (First output event)",563,"ms","563 ms",5,"\u0001","minmax","Claude Code · CLI start-up, one-word prompt, 5 runs",[519,726],"\u0001","\u0001","\u0001"],["routing-overhead","cli-startup-tax","First model output","Claude Code · Claude Haiku 4.5","cli-startup-tax/First model output","CLI start-up tax on a one-word answer (First model output)",1461,"ms","1,461 ms",5,"\u0001","minmax","Claude Code · CLI start-up, one-word prompt, 5 runs",[1206,2308],"\u0001","\u0001","\u0001"],["routing-overhead","cli-startup-tax","Total wall time","Claude Code · Claude Haiku 4.5","cli-startup-tax/Total wall time","CLI start-up tax on a one-word answer (Total wall time)",2529,"ms","2,529 ms",5,"\u0001","minmax","Claude Code · CLI start-up, one-word prompt, 5 runs",[2273,3382],"\u0001","\u0001","\u0001"],["routing-overhead","cli-startup-input-tokens","Input tokens per call","Claude Code · Claude Haiku 4.5","cli-startup-input-tokens","Input tokens a CLI sends for a one-word answer",6761,"tokens","6,761",5,"\u0001","\u0001","Claude Code · CLI start-up, one-word prompt, 5 runs","\u0001","\u0001","\u0001","\u0001"],["cost-thought-experiments","repriced-cost-per-resolved","Repriced cost per resolved instance","Claude Haiku 4.5","repriced-cost-per-resolved","Thought experiment: the same tokens at other list prices",1.745,"usd","$1.75","\u0001","\u0001","\u0001","calculation: Agent’s recorded tokens at this model’s list price","\u0001",true,"\u0001","\u0001"],["cost-thought-experiments","cost-per-resolved-agent-vs-panel","Cost per resolved instance","Claude 4.5 Haiku (high)","cost-per-resolved-agent-vs-panel","Recorded cost per resolved instance: Agent vs the public panel",0.479,"usd","$0.48",25,"\u0001","\u0001","effort high · public mini-SWE-agent v2 run, same instances","\u0001","\u0001","\u0001","\u0001"],["cost-thought-experiments","prompt-cache-savings","With caching (as recorded)","Claude Haiku 4.5","prompt-cache-savings/With caching (as recorded)","Thought experiment: what prompt caching saved (With caching (as recorded))",43.61,"usd","$43.61","\u0001","\u0001","\u0001","calculation: Agent’s recorded tokens at this model’s list price","\u0001",true,"\u0001","\u0001"],["cost-thought-experiments","prompt-cache-savings","Without caching","Claude Haiku 4.5","prompt-cache-savings/Without caching","Thought experiment: what prompt caching saved (Without caching)",171.66,"usd","$171.66","\u0001","\u0001","\u0001","calculation: Agent’s recorded tokens at this model’s list price","\u0001",true,"\u0001","\u0001"],["single-call-vs-agent-loop","agent-loop-pass-rate","Strict pass","Claude Haiku 4.5 (single call) · Claude Code","agent-loop-pass-rate","Strict pass rate: single call vs agent loop on eight hard tasks",0.4583,"rate","46% (11/24)",24,[0.2789,0.6493],"ci95","Claude Code · single call","\u0001","\u0001","higher","\u0001"],["single-call-vs-agent-loop","agent-loop-pass-rate","Strict pass","Claude Haiku 4.5 (agent loop) · Claude Code","agent-loop-pass-rate","Strict pass rate: single call vs agent loop on eight hard tasks",0.5417,"rate","54% (13/24)",24,[0.3507,0.7211],"ci95","Claude Code · agent loop","\u0001","\u0001","higher","\u0001"],["single-call-vs-agent-loop","agent-loop-by-task","Claude Haiku 4.5 (single call) · Claude Code","Interval merge fix","agent-loop-by-task/Interval merge fix","Strict passes per task: single call vs agent loop: Interval merge fix",1,"rate","100% (3/3)",3,[0.4385,1],"ci95","Claude Code · single call","\u0001","\u0001","higher","\u0001"],["single-call-vs-agent-loop","agent-loop-by-task","Claude Haiku 4.5 (single call) · Claude Code","DST day-length fix","agent-loop-by-task/DST day-length fix","Strict passes per task: single call vs agent loop: DST day-length fix",0.3333,"rate","33% (1/3)",3,[0.0615,0.7923],"ci95","Claude Code · single call","\u0001","\u0001","higher","\u0001"],["single-call-vs-agent-loop","agent-loop-by-task","Claude Haiku 4.5 (single call) · Claude Code","CSV parser","agent-loop-by-task/CSV parser","Strict passes per task: single call vs agent loop: CSV parser",0.6667,"rate","67% (2/3)",3,[0.2077,0.9385],"ci95","Claude Code · single call","\u0001","\u0001","higher","\u0001"],["single-call-vs-agent-loop","agent-loop-by-task","Claude Haiku 4.5 (single call) · Claude Code","Event-loop order","agent-loop-by-task/Event-loop order","Strict passes per task: single call vs agent loop: Event-loop order",0,"rate","0% (0/3)",3,[0,0.5615],"ci95","Claude Code · single call","\u0001","\u0001","higher","\u0001"],["single-call-vs-agent-loop","agent-loop-by-task","Claude Haiku 4.5 (single call) · Claude Code","Room schedule","agent-loop-by-task/Room schedule","Strict passes per task: single call vs agent loop: Room schedule",0,"rate","0% (0/3)",3,[0,0.5615],"ci95","Claude Code · single call","\u0001","\u0001","higher","\u0001"],["single-call-vs-agent-loop","agent-loop-by-task","Claude Haiku 4.5 (single call) · Claude Code","SemVer regex","agent-loop-by-task/SemVer regex","Strict passes per task: single call vs agent loop: SemVer regex",1,"rate","100% (3/3)",3,[0.4385,1],"ci95","Claude Code · single call","\u0001","\u0001","higher","\u0001"],["single-call-vs-agent-loop","agent-loop-by-task","Claude Haiku 4.5 (single call) · Claude Code","Money refactor","agent-loop-by-task/Money refactor","Strict passes per task: single call vs agent loop: Money refactor",0.6667,"rate","67% (2/3)",3,[0.2077,0.9385],"ci95","Claude Code · single call","\u0001","\u0001","higher","\u0001"],["single-call-vs-agent-loop","agent-loop-by-task","Claude Haiku 4.5 (single call) · Claude Code","SQL report","agent-loop-by-task/SQL report","Strict passes per task: single call vs agent loop: SQL report",0,"rate","0% (0/3)",3,[0,0.5615],"ci95","Claude Code · single call","\u0001","\u0001","higher","\u0001"],["single-call-vs-agent-loop","agent-loop-by-task","Claude Haiku 4.5 (agent loop) · Claude Code","Interval merge fix","agent-loop-by-task/Interval merge fix","Strict passes per task: single call vs agent loop: Interval merge fix",0.6667,"rate","67% (2/3)",3,[0.2077,0.9385],"ci95","Claude Code · agent loop","\u0001","\u0001","higher","\u0001"],["single-call-vs-agent-loop","agent-loop-by-task","Claude Haiku 4.5 (agent loop) · Claude Code","DST day-length fix","agent-loop-by-task/DST day-length fix","Strict passes per task: single call vs agent loop: DST day-length fix",0.6667,"rate","67% (2/3)",3,[0.2077,0.9385],"ci95","Claude Code · agent loop","\u0001","\u0001","higher","\u0001"],["single-call-vs-agent-loop","agent-loop-by-task","Claude Haiku 4.5 (agent loop) · Claude Code","CSV parser","agent-loop-by-task/CSV parser","Strict passes per task: single call vs agent loop: CSV parser",0.6667,"rate","67% (2/3)",3,[0.2077,0.9385],"ci95","Claude Code · agent loop","\u0001","\u0001","higher","\u0001"],["single-call-vs-agent-loop","agent-loop-by-task","Claude Haiku 4.5 (agent loop) · Claude Code","Event-loop order","agent-loop-by-task/Event-loop order","Strict passes per task: single call vs agent loop: Event-loop order",1,"rate","100% (3/3)",3,[0.4385,1],"ci95","Claude Code · agent loop","\u0001","\u0001","higher","\u0001"],["single-call-vs-agent-loop","agent-loop-by-task","Claude Haiku 4.5 (agent loop) · Claude Code","Room schedule","agent-loop-by-task/Room schedule","Strict passes per task: single call vs agent loop: Room schedule",0.6667,"rate","67% (2/3)",3,[0.2077,0.9385],"ci95","Claude Code · agent loop","\u0001","\u0001","higher","\u0001"],["single-call-vs-agent-loop","agent-loop-by-task","Claude Haiku 4.5 (agent loop) · Claude Code","SemVer regex","agent-loop-by-task/SemVer regex","Strict passes per task: single call vs agent loop: SemVer regex",0.6667,"rate","67% (2/3)",3,[0.2077,0.9385],"ci95","Claude Code · agent loop","\u0001","\u0001","higher","\u0001"],["single-call-vs-agent-loop","agent-loop-by-task","Claude Haiku 4.5 (agent loop) · Claude Code","Money refactor","agent-loop-by-task/Money refactor","Strict passes per task: single call vs agent loop: Money refactor",0,"rate","0% (0/3)",3,[0,0.5615],"ci95","Claude Code · agent loop","\u0001","\u0001","higher","\u0001"],["single-call-vs-agent-loop","agent-loop-by-task","Claude Haiku 4.5 (agent loop) · Claude Code","SQL report","agent-loop-by-task/SQL report","Strict passes per task: single call vs agent loop: SQL report",0,"rate","0% (0/3)",3,[0,0.5615],"ci95","Claude Code · agent loop","\u0001","\u0001","higher","\u0001"],["single-call-vs-agent-loop","agent-loop-total-time","Total time per attempt","Claude Haiku 4.5 (single call) · Claude Code","agent-loop-total-time","Total time per attempt: single call vs agent loop",39.01,"seconds","39.0 s",24,"\u0001","minmax","Claude Code · single call",[15.27,75.13],"\u0001","\u0001","\u0001"],["single-call-vs-agent-loop","agent-loop-total-time","Total time per attempt","Claude Haiku 4.5 (agent loop) · Claude Code","agent-loop-total-time","Total time per attempt: single call vs agent loop",56.77,"seconds","56.8 s",24,"\u0001","minmax","Claude Code · agent loop",[24.53,223.7],"\u0001","\u0001","\u0001"],["single-call-vs-agent-loop","agent-loop-tokens","Input tokens (cache reads included)","Claude Haiku 4.5 (single call) · Claude Code","agent-loop-tokens/Input tokens (cache reads included)","Tokens per attempt: single call vs agent loop (Input tokens (cache reads included))",3941,"tokens","3,941",24,"\u0001","minmax","Claude Code · single call",[3879,4221],"\u0001","\u0001","\u0001"],["single-call-vs-agent-loop","agent-loop-tokens","Input tokens (cache reads included)","Claude Haiku 4.5 (agent loop) · Claude Code","agent-loop-tokens/Input tokens (cache reads included)","Tokens per attempt: single call vs agent loop (Input tokens (cache reads included))",71691,"tokens","71,691",24,"\u0001","minmax","Claude Code · agent loop",[41732,516306],"\u0001","\u0001","\u0001"],["single-call-vs-agent-loop","agent-loop-tokens","Output tokens","Claude Haiku 4.5 (single call) · Claude Code","agent-loop-tokens/Output tokens","Tokens per attempt: single call vs agent loop (Output tokens)",5064,"tokens","5,064",24,"\u0001","minmax","Claude Code · single call",[1899,9321],"\u0001","\u0001","\u0001"],["single-call-vs-agent-loop","agent-loop-tokens","Output tokens","Claude Haiku 4.5 (agent loop) · Claude Code","agent-loop-tokens/Output tokens","Tokens per attempt: single call vs agent loop (Output tokens)",7912,"tokens","7,912",24,"\u0001","minmax","Claude Code · agent loop",[2541,20654],"\u0001","\u0001","\u0001"],["single-call-vs-agent-loop","agent-loop-tool-calls","Tool calls per attempt","Claude Haiku 4.5 (agent loop) · Claude Code","agent-loop-tool-calls","Tool calls per agent-loop attempt",3,"count","3",24,"\u0001","minmax","Claude Code · agent loop",[2,18],"\u0001","\u0001","\u0001"],["single-call-vs-agent-loop","agent-loop-cost-per-pass","Cost per strict pass","Claude Haiku 4.5 (single call) · Claude Code","agent-loop-cost-per-pass","List-price cost per strict pass: single call vs agent loop (calculation)",0.0672,"usd","$0.067",24,"\u0001","\u0001","Claude Code · single call","\u0001",true,"\u0001","\u0001"],["single-call-vs-agent-loop","agent-loop-cost-per-pass","Cost per strict pass","Claude Haiku 4.5 (agent loop) · Claude Code","agent-loop-cost-per-pass","List-price cost per strict pass: single call vs agent loop (calculation)",0.14225,"usd","$0.14",24,"\u0001","\u0001","Claude Code · agent loop","\u0001",true,"\u0001","\u0001"],["haiku-thinking-on-off","haiku-thinking-router-exact","Exact decisions (every scored question right)","Claude Haiku 4.5 (thinking off) · Claude Code","haiku-thinking-router-exact/Exact decisions (every scored question right)","Haiku thinking study: typed routing decisions answered exactly right (Exact decisions (every scored question right))",0.8659,"rate","87% (71/82)",82,[0.7755,0.9234],"ci95","Claude Code · thinking off · typed routing decisions, thinking on vs off","\u0001","\u0001","higher","\u0001"],["haiku-thinking-on-off","haiku-thinking-router-exact","Exact decisions (every scored question right)","Claude Haiku 4.5 (thinking on) · Claude Code","haiku-thinking-router-exact/Exact decisions (every scored question right)","Haiku thinking study: typed routing decisions answered exactly right (Exact decisions (every scored question right))",0.8902,"rate","89% (73/82)",82,[0.8044,0.9412],"ci95","Claude Code · thinking on · typed routing decisions, thinking on vs off","\u0001","\u0001","higher","\u0001"],["haiku-thinking-on-off","haiku-thinking-router-exact","Per-question accuracy","Claude Haiku 4.5 (thinking off) · Claude Code","haiku-thinking-router-exact/Per-question accuracy","Haiku thinking study: typed routing decisions answered exactly right (Per-question accuracy)",0.9124,"rate","91% (177/194)",194,[0.8642,0.9446],"ci95","Claude Code · thinking off · typed routing decisions, thinking on vs off","\u0001","\u0001","higher","\u0001"],["haiku-thinking-on-off","haiku-thinking-router-exact","Per-question accuracy","Claude Haiku 4.5 (thinking on) · Claude Code","haiku-thinking-router-exact/Per-question accuracy","Haiku thinking study: typed routing decisions answered exactly right (Per-question accuracy)",0.9433,"rate","94% (183/194)",194,[0.9013,0.968],"ci95","Claude Code · thinking on · typed routing decisions, thinking on vs off","\u0001","\u0001","higher","\u0001"],["haiku-thinking-on-off","haiku-thinking-router-latency","Wall time (CLI)","Claude Haiku 4.5 (thinking off) · Claude Code","haiku-thinking-router-latency/Wall time (CLI)","Haiku thinking study: time per routing decision (Wall time (CLI))",4.66,"seconds","4.66 s",82,"\u0001","p50-p95","Claude Code · thinking off · typed routing decisions, thinking on vs off",[4.66,8.18],"\u0001","\u0001","\u0001"],["haiku-thinking-on-off","haiku-thinking-router-latency","Wall time (CLI)","Claude Haiku 4.5 (thinking on) · Claude Code","haiku-thinking-router-latency/Wall time (CLI)","Haiku thinking study: time per routing decision (Wall time (CLI))",12.54,"seconds","12.5 s",82,"\u0001","p50-p95","Claude Code · thinking on · typed routing decisions, thinking on vs off",[12.54,34.48],"\u0001","\u0001","\u0001"],["haiku-thinking-on-off","haiku-thinking-router-latency","Model time (API)","Claude Haiku 4.5 (thinking off) · Claude Code","haiku-thinking-router-latency/Model time (API)","Haiku thinking study: time per routing decision (Model time (API))",3.79,"seconds","3.79 s",82,"\u0001","p50-p95","Claude Code · thinking off · typed routing decisions, thinking on vs off",[3.79,7.43],"\u0001","\u0001","\u0001"],["haiku-thinking-on-off","haiku-thinking-router-latency","Model time (API)","Claude Haiku 4.5 (thinking on) · Claude Code","haiku-thinking-router-latency/Model time (API)","Haiku thinking study: time per routing decision (Model time (API))",10.51,"seconds","10.5 s",82,"\u0001","p50-p95","Claude Code · thinking on · typed routing decisions, thinking on vs off",[10.51,32.13],"\u0001","\u0001","\u0001"],["haiku-thinking-on-off","haiku-thinking-router-tokens","Thinking tokens","Claude Haiku 4.5 (thinking off) · Claude Code","haiku-thinking-router-tokens/Thinking tokens","Haiku thinking study: thinking and visible output tokens per routing decision (Thinking tokens)",0,"tokens","0",82,"\u0001","\u0001","Claude Code · thinking off · typed routing decisions, thinking on vs off","\u0001","\u0001","\u0001","\u0001"],["haiku-thinking-on-off","haiku-thinking-router-tokens","Thinking tokens","Claude Haiku 4.5 (thinking on) · Claude Code","haiku-thinking-router-tokens/Thinking tokens","Haiku thinking study: thinking and visible output tokens per routing decision (Thinking tokens)",1101,"tokens","1,101",82,"\u0001","\u0001","Claude Code · thinking on · typed routing decisions, thinking on vs off","\u0001","\u0001","\u0001","\u0001"],["haiku-thinking-on-off","haiku-thinking-router-tokens","Visible output tokens","Claude Haiku 4.5 (thinking off) · Claude Code","haiku-thinking-router-tokens/Visible output tokens","Haiku thinking study: thinking and visible output tokens per routing decision (Visible output tokens)",366,"tokens","366",82,"\u0001","\u0001","Claude Code · thinking off · typed routing decisions, thinking on vs off","\u0001","\u0001","\u0001","\u0001"],["haiku-thinking-on-off","haiku-thinking-router-tokens","Visible output tokens","Claude Haiku 4.5 (thinking on) · Claude Code","haiku-thinking-router-tokens/Visible output tokens","Haiku thinking study: thinking and visible output tokens per routing decision (Visible output tokens)",318,"tokens","318",82,"\u0001","\u0001","Claude Code · thinking on · typed routing decisions, thinking on vs off","\u0001","\u0001","\u0001","\u0001"],["haiku-thinking-on-off","haiku-thinking-router-cost","Cost per 1,000 decisions","Claude Haiku 4.5 (thinking off) · Claude Code","haiku-thinking-router-cost","Haiku thinking study: list-price cost per 1,000 routing decisions (calculation)",3.364,"usd","$3.36",82,"\u0001","\u0001","Claude Code · thinking off · typed routing decisions, thinking on vs off","\u0001",true,"\u0001","\u0001"],["haiku-thinking-on-off","haiku-thinking-router-cost","Cost per 1,000 decisions","Claude Haiku 4.5 (thinking on) · Claude Code","haiku-thinking-router-cost","Haiku thinking study: list-price cost per 1,000 routing decisions (calculation)",8.924,"usd","$8.92",82,"\u0001","\u0001","Claude Code · thinking on · typed routing decisions, thinking on vs off","\u0001",true,"\u0001","\u0001"],["haiku-thinking-on-off","haiku-thinking-hard-pass","Strict pass","Claude Haiku 4.5 (thinking off) · Claude Code","haiku-thinking-hard-pass/Strict pass","Haiku thinking study: pass rate on eight hard tasks (Strict pass)",0.1667,"rate","17% (4/24)",24,[0.0668,0.3585],"ci95","Claude Code · thinking off · eight hard validated tasks, thinking on vs off","\u0001","\u0001","higher","\u0001"],["haiku-thinking-on-off","haiku-thinking-hard-pass","Strict pass","Claude Haiku 4.5 (thinking on) · Claude Code","haiku-thinking-hard-pass/Strict pass","Haiku thinking study: pass rate on eight hard tasks (Strict pass)",0.4583,"rate","46% (11/24)",24,[0.2789,0.6493],"ci95","Claude Code · thinking on · eight hard validated tasks, thinking on vs off","\u0001","\u0001","higher","\u0001"],["haiku-thinking-on-off","haiku-thinking-hard-pass","Lenient (format misses counted)","Claude Haiku 4.5 (thinking off) · Claude Code","haiku-thinking-hard-pass/Lenient (format misses counted)","Haiku thinking study: pass rate on eight hard tasks (Lenient (format misses counted))",0.1667,"rate","17% (4/24)",24,[0.0668,0.3585],"ci95","Claude Code · thinking off · eight hard validated tasks, thinking on vs off","\u0001","\u0001","higher","\u0001"],["haiku-thinking-on-off","haiku-thinking-hard-pass","Lenient (format misses counted)","Claude Haiku 4.5 (thinking on) · Claude Code","haiku-thinking-hard-pass/Lenient (format misses counted)","Haiku thinking study: pass rate on eight hard tasks (Lenient (format misses counted))",0.6667,"rate","67% (16/24)",24,[0.4671,0.8203],"ci95","Claude Code · thinking on · eight hard validated tasks, thinking on vs off","\u0001","\u0001","higher","\u0001"],["haiku-thinking-on-off","haiku-thinking-hard-time","Total time per call","Claude Haiku 4.5 (thinking off) · Claude Code","haiku-thinking-hard-time","Haiku thinking study: total time per call on hard tasks",2.95,"seconds","2.95 s",24,"\u0001","minmax","Claude Code · thinking off · eight hard validated tasks, thinking on vs off",[1.7,13.01],"\u0001","\u0001","\u0001"],["haiku-thinking-on-off","haiku-thinking-hard-time","Total time per call","Claude Haiku 4.5 (thinking on) · Claude Code","haiku-thinking-hard-time","Haiku thinking study: total time per call on hard tasks",39.01,"seconds","39.0 s",24,"\u0001","minmax","Claude Code · thinking on · eight hard validated tasks, thinking on vs off",[15.27,75.13],"\u0001","\u0001","\u0001"],["haiku-thinking-on-off","\u0001","\u0001","\u0001","stat:haiku-thinking-hard-reasoning-on","Claude Haiku 4.5 (thinking on): median reasoning tokens per hard-task call",4556,"tokens","4,556",24,"\u0001","\u0001","thinking on","\u0001","\u0001","\u0001","haiku-thinking-hard-reasoning-on"],["haiku-thinking-on-off","\u0001","\u0001","\u0001","stat:haiku-thinking-hard-cost-per-pass-off","Claude Haiku 4.5 (thinking off): list-price cost per strict pass on hard tasks (calculation)",0.03654,"usd","$0.0365",24,"\u0001","\u0001","thinking off","\u0001",true,"\u0001","haiku-thinking-hard-cost-per-pass-off"],["haiku-thinking-on-off","\u0001","\u0001","\u0001","stat:haiku-thinking-hard-cost-per-pass-on","Claude Haiku 4.5 (thinking on): list-price cost per strict pass on hard tasks (calculation)",0.0672,"usd","$0.0672",24,"\u0001","\u0001","thinking on","\u0001",true,"\u0001","haiku-thinking-hard-cost-per-pass-on"],["json-schema-vs-instructions","structured-output-pass-rate","Strict pass: the whole reply is the right JSON","Claude Haiku 4.5 (instructions) · Claude Code","structured-output-pass-rate/Strict pass: the whole reply is the right JSON","Does a JSON schema raise the pass rate? Instructions vs schema mode (Strict pass: the whole reply is the right JSON)",0,"rate","0% (0/24)",24,[0,0.138],"ci95","Claude Code · instructions","\u0001",true,"higher","\u0001"],["json-schema-vs-instructions","structured-output-pass-rate","Strict pass: the whole reply is the right JSON","Claude Haiku 4.5 (JSON schema) · Claude Code","structured-output-pass-rate/Strict pass: the whole reply is the right JSON","Does a JSON schema raise the pass rate? Instructions vs schema mode (Strict pass: the whole reply is the right JSON)",0.75,"rate","75% (18/24)",24,[0.551,0.88],"ci95","Claude Code · JSON schema","\u0001",true,"higher","\u0001"],["json-schema-vs-instructions","structured-output-pass-rate","Right answer in any format (strict pass or format miss)","Claude Haiku 4.5 (instructions) · Claude Code","structured-output-pass-rate/Right answer in any format (strict pass or format miss)","Does a JSON schema raise the pass rate? Instructions vs schema mode (Right answer in any format (strict pass or format miss))",0.7083,"rate","71% (17/24)",24,[0.5083,0.8509],"ci95","Claude Code · instructions","\u0001",true,"higher","\u0001"],["json-schema-vs-instructions","structured-output-pass-rate","Right answer in any format (strict pass or format miss)","Claude Haiku 4.5 (JSON schema) · Claude Code","structured-output-pass-rate/Right answer in any format (strict pass or format miss)","Does a JSON schema raise the pass rate? Instructions vs schema mode (Right answer in any format (strict pass or format miss))",0.75,"rate","75% (18/24)",24,[0.551,0.88],"ci95","Claude Code · JSON schema","\u0001",true,"higher","\u0001"],["json-schema-vs-instructions","structured-output-outcomes","Strict pass","Claude Haiku 4.5 (instructions) · Claude Code","structured-output-outcomes/Strict pass","What each call produced: strict pass, format miss, wrong values or error (Strict pass)",0,"count","0",24,"\u0001","\u0001","Claude Code · instructions","\u0001","\u0001","\u0001","\u0001"],["json-schema-vs-instructions","structured-output-outcomes","Strict pass","Claude Haiku 4.5 (JSON schema) · Claude Code","structured-output-outcomes/Strict pass","What each call produced: strict pass, format miss, wrong values or error (Strict pass)",18,"count","18",24,"\u0001","\u0001","Claude Code · JSON schema","\u0001","\u0001","\u0001","\u0001"],["json-schema-vs-instructions","structured-output-outcomes","Format miss","Claude Haiku 4.5 (instructions) · Claude Code","structured-output-outcomes/Format miss","What each call produced: strict pass, format miss, wrong values or error (Format miss)",17,"count","17",24,"\u0001","\u0001","Claude Code · instructions","\u0001","\u0001","\u0001","\u0001"],["json-schema-vs-instructions","structured-output-outcomes","Format miss","Claude Haiku 4.5 (JSON schema) · Claude Code","structured-output-outcomes/Format miss","What each call produced: strict pass, format miss, wrong values or error (Format miss)",0,"count","0",24,"\u0001","\u0001","Claude Code · JSON schema","\u0001","\u0001","\u0001","\u0001"],["json-schema-vs-instructions","structured-output-outcomes","Wrong values","Claude Haiku 4.5 (instructions) · Claude Code","structured-output-outcomes/Wrong values","What each call produced: strict pass, format miss, wrong values or error (Wrong values)",7,"count","7",24,"\u0001","\u0001","Claude Code · instructions","\u0001","\u0001","\u0001","\u0001"],["json-schema-vs-instructions","structured-output-outcomes","Wrong values","Claude Haiku 4.5 (JSON schema) · Claude Code","structured-output-outcomes/Wrong values","What each call produced: strict pass, format miss, wrong values or error (Wrong values)",6,"count","6",24,"\u0001","\u0001","Claude Code · JSON schema","\u0001","\u0001","\u0001","\u0001"],["json-schema-vs-instructions","structured-output-outcomes","Error","Claude Haiku 4.5 (instructions) · Claude Code","structured-output-outcomes/Error","What each call produced: strict pass, format miss, wrong values or error (Error)",0,"count","0",24,"\u0001","\u0001","Claude Code · instructions","\u0001","\u0001","\u0001","\u0001"],["json-schema-vs-instructions","structured-output-outcomes","Error","Claude Haiku 4.5 (JSON schema) · Claude Code","structured-output-outcomes/Error","What each call produced: strict pass, format miss, wrong values or error (Error)",0,"count","0",24,"\u0001","\u0001","Claude Code · JSON schema","\u0001","\u0001","\u0001","\u0001"],["json-schema-vs-instructions","structured-output-time","Median time per call (the three prompts pooled)","Claude Haiku 4.5 (instructions) · Claude Code","structured-output-time","Time per call, instructions vs schema mode",9.52,"seconds","9.52 s",24,"\u0001","minmax","Claude Code · instructions",[5.67,17],"\u0001","\u0001","\u0001"],["json-schema-vs-instructions","structured-output-time","Median time per call (the three prompts pooled)","Claude Haiku 4.5 (JSON schema) · Claude Code","structured-output-time","Time per call, instructions vs schema mode",8.46,"seconds","8.46 s",24,"\u0001","minmax","Claude Code · JSON schema",[5.9,12.23],"\u0001","\u0001","\u0001"],["json-schema-vs-instructions","structured-output-tokens","Median output tokens per call","Claude Haiku 4.5 (instructions) · Claude Code","structured-output-tokens/Median output tokens per call","Output and reasoning tokens per call, instructions vs schema mode (Median output tokens per call)",1128,"tokens","1,128",24,"\u0001","\u0001","Claude Code · instructions","\u0001","\u0001","\u0001","\u0001"],["json-schema-vs-instructions","structured-output-tokens","Median output tokens per call","Claude Haiku 4.5 (JSON schema) · Claude Code","structured-output-tokens/Median output tokens per call","Output and reasoning tokens per call, instructions vs schema mode (Median output tokens per call)",1036,"tokens","1,036",24,"\u0001","\u0001","Claude Code · JSON schema","\u0001","\u0001","\u0001","\u0001"],["prompt-cache-break-even","cache-break-even-reads","1-hour write (2× input), whole prefix new","Claude Haiku 4.5","cache-break-even-reads/1-hour write (2× input), whole prefix new","Reuses before a cached prefix costs less, by model and write type (calculation) (1-hour write (2× input), whole prefix new)",1.11,"score","1.11","\u0001","\u0001","\u0001","calculation: Agent’s recorded tokens at this model’s list price","\u0001",true,"\u0001","\u0001"],["prompt-cache-break-even","cache-break-even-reads","1-hour write, pooled n = 6 session share, 19% already cached (as recorded)","Claude Haiku 4.5","cache-break-even-reads/1-hour write, pooled n = 6 session share, 19% already cached (as recorded)","Reuses before a cached prefix costs less, by model and write type (calculation) (1-hour write, pooled n = 6 session share, 19% already cached (as recorded))",0.72,"score","0.72","\u0001","\u0001","\u0001","calculation: Agent’s recorded tokens at this model’s list price","\u0001",true,"\u0001","\u0001"],["prompt-cache-break-even","cache-break-even-reads","5-minute write (1.25× input, an assumption)","Claude Haiku 4.5","cache-break-even-reads/5-minute write (1.25× input, an assumption)","Reuses before a cached prefix costs less, by model and write type (calculation) (5-minute write (1.25× input, an assumption))",0.28,"score","0.28","\u0001","\u0001","\u0001","calculation: Agent’s recorded tokens at this model’s list price","\u0001",true,"\u0001","\u0001"],["routing-holdout","routing-holdout-exact","Exact rate","Claude Haiku 4.5 · Claude Code","routing-holdout-exact","Unseen routing decisions answered exactly right",0.7857,"rate","79% (44/56)",56,[0.6618,0.8729],"ci95","Claude Code","\u0001","\u0001","higher","\u0001"],["routing-holdout","routing-holdout-key-accuracy","Key accuracy","Claude Haiku 4.5 · Claude Code","routing-holdout-key-accuracy","Per-question accuracy on unseen decisions",0.816,"rate","82% (102/125)",125,[0.739,0.8741],"ci95","Claude Code","\u0001","\u0001","higher","\u0001"],["routing-holdout","routing-holdout-by-purpose","Claude Haiku 4.5 · Claude Code","Failure class","routing-holdout-by-purpose/Failure class","Exact rate on unseen decisions, by decision type: Failure class",0.9286,"rate","93% (13/14)",14,[0.6853,0.9873],"ci95","Claude Code","\u0001","\u0001","higher","\u0001"],["routing-holdout","routing-holdout-by-purpose","Claude Haiku 4.5 · Claude Code","Message intent","routing-holdout-by-purpose/Message intent","Exact rate on unseen decisions, by decision type: Message intent",0.9286,"rate","93% (13/14)",14,[0.6853,0.9873],"ci95","Claude Code","\u0001","\u0001","higher","\u0001"],["routing-holdout","routing-holdout-by-purpose","Claude Haiku 4.5 · Claude Code","Is it a rule?","routing-holdout-by-purpose/Is it a rule?","Exact rate on unseen decisions, by decision type: Is it a rule?",0.9286,"rate","93% (13/14)",14,[0.6853,0.9873],"ci95","Claude Code","\u0001","\u0001","higher","\u0001"],["routing-holdout","routing-holdout-by-purpose","Claude Haiku 4.5 · Claude Code","Context shape","routing-holdout-by-purpose/Context shape","Exact rate on unseen decisions, by decision type: Context shape",0.3571,"rate","36% (5/14)",14,[0.1634,0.6124],"ci95","Claude Code","\u0001","\u0001","higher","\u0001"],["routing-holdout","routing-holdout-tuned-vs-unseen","Tuned set (routing-jev-vs-llm)","Claude Haiku 4.5 · Claude Code","routing-holdout-tuned-vs-unseen/Tuned set (routing-jev-vs-llm)","Tuned case set vs unseen holdout: exact rate per router (Tuned set (routing-jev-vs-llm))",0.8902,"rate","89% (73/82)",82,[0.8044,0.9412],"ci95","Claude Code","\u0001","\u0001","higher","\u0001"],["routing-holdout","routing-holdout-tuned-vs-unseen","Unseen holdout","Claude Haiku 4.5 · Claude Code","routing-holdout-tuned-vs-unseen/Unseen holdout","Tuned case set vs unseen holdout: exact rate per router (Unseen holdout)",0.7857,"rate","79% (44/56)",56,[0.6618,0.8729],"ci95","Claude Code","\u0001","\u0001","higher","\u0001"],["routing-holdout","routing-holdout-latency","Wall time","Claude Haiku 4.5 · Claude Code","routing-holdout-latency/Wall time","Time per routing decision, by route (Wall time)",9.444,"seconds","9.44 s",56,"\u0001","p50-p95","Claude Code",[9.444,25.413],"\u0001","\u0001","\u0001"],["routing-holdout","routing-holdout-latency","Model time (API, CLI-reported)","Claude Haiku 4.5 · Claude Code","routing-holdout-latency/Model time (API, CLI-reported)","Time per routing decision, by route (Model time (API, CLI-reported))",7.522,"seconds","7.52 s",56,"\u0001","p50-p95","Claude Code",[7.522,23.913],"\u0001","\u0001","\u0001"],["routing-holdout","routing-holdout-cost-per-1000","Cost","Claude Haiku 4.5 · Claude Code","routing-holdout-cost-per-1000","Cost per 1,000 unseen routing decisions",7.129,"usd","$7.13",56,"\u0001","\u0001","Claude Code","\u0001",true,"\u0001","\u0001"],["routing-holdout","\u0001","\u0001","\u0001","stat:holdout-gap-claude-haiku","Claude Haiku 4.5 · Claude Code: holdout minus tuned-set exact rate",-0.1045,"rate","−10.5 points",56,"\u0001","\u0001","","\u0001",true,"\u0001","holdout-gap-claude-haiku"],["thinking-token-bill","thinking-bill-share","Median call","Claude Haiku 4.5 · Claude Code","thinking-bill-share","Reasoning share of output tokens per call on hard tasks (calculation)",91.68,"percent","91.7%",24,"\u0001","minmax","Claude Code",[76.46,99.27],true,"none","\u0001"],["thinking-token-bill","thinking-bill-cost-per-call","Reasoning (output tokens)","Claude Haiku 4.5 · Claude Code","thinking-bill-cost-per-call/Reasoning (output tokens)","List-price cost per call: reasoning, remaining output and input (calculation) (Reasoning (output tokens))",0.024492,"usd","$0.024",24,"\u0001","\u0001","Claude Code","\u0001",true,"none","\u0001"],["thinking-token-bill","thinking-bill-cost-per-call","Remaining output (visible-answer estimate)","Claude Haiku 4.5 · Claude Code","thinking-bill-cost-per-call/Remaining output (visible-answer estimate)","List-price cost per call: reasoning, remaining output and input (calculation) (Remaining output (visible-answer estimate))",0.001795,"usd","$0.0018",24,"\u0001","\u0001","Claude Code","\u0001",true,"none","\u0001"],["thinking-token-bill","thinking-bill-cost-per-call","Input (prompt, cache priced)","Claude Haiku 4.5 · Claude Code","thinking-bill-cost-per-call/Input (prompt, cache priced)","List-price cost per call: reasoning, remaining output and input (calculation) (Input (prompt, cache priced))",0.00451,"usd","$0.0045",24,"\u0001","\u0001","Claude Code","\u0001",true,"none","\u0001"],["thinking-token-bill","thinking-bill-short-vs-hard","Eight hard tasks","Claude Haiku 4.5 · Claude Code","thinking-bill-short-vs-hard/Eight hard tasks","Reasoning share on short tasks vs hard tasks (calculation) (Eight hard tasks)",91.68,"percent","91.7%",24,"\u0001","minmax","Claude Code",[76.46,99.27],true,"none","\u0001"],["thinking-token-bill","thinking-bill-short-vs-hard","Five short tasks","Claude Haiku 4.5 · Claude Code","thinking-bill-short-vs-hard/Five short tasks","Reasoning share on short tasks vs hard tasks (calculation) (Five short tasks)",90.19,"percent","90.2%",15,"\u0001","minmax","Claude Code",[73.1,97.59],true,"none","\u0001"],["llm-speed-anatomy","speed-anatomy-first-text","Time to first text","Claude Haiku 4.5 · Claude Code","speed-anatomy-first-text","Time to first text: a 250-line answer, six models",4,"seconds","4.00 s",4,"\u0001","minmax","Claude Code",[2.84,6.38],"\u0001","\u0001","\u0001"],["llm-speed-anatomy","speed-anatomy-output-speed","Visible tokens per second","Claude Haiku 4.5 · Claude Code","speed-anatomy-output-speed","Output speed after the first text: visible tokens per second (calculation)",153.2,"tokens","153",4,"\u0001","minmax","Claude Code",[152.6,216.1],true,"\u0001","\u0001"],["llm-speed-anatomy","speed-anatomy-chars-per-second","Characters per second","Claude Haiku 4.5 · Claude Code","speed-anatomy-chars-per-second","Output speed in characters per second after the first text (calculation)",547,"count","547",3,"\u0001","minmax","Claude Code",[546,548],true,"\u0001","\u0001"],["llm-speed-anatomy","speed-anatomy-prompt-size","Claude Haiku 4.5 · Claude Code","1k","speed-anatomy-prompt-size/1k","Time to first text as the prompt grows: 1k",1.93,"seconds","1.93 s",3,"\u0001","minmax","Claude Code",[1.85,2.04],true,"\u0001","\u0001"],["llm-speed-anatomy","speed-anatomy-prompt-size","Claude Haiku 4.5 · Claude Code","16k","speed-anatomy-prompt-size/16k","Time to first text as the prompt grows: 16k",2.27,"seconds","2.27 s",3,"\u0001","minmax","Claude Code",[2.22,2.47],true,"\u0001","\u0001"],["llm-speed-anatomy","speed-anatomy-prompt-size","Claude Haiku 4.5 · Claude Code","64k","speed-anatomy-prompt-size/64k","Time to first text as the prompt grows: 64k",2.78,"seconds","2.78 s",3,"\u0001","minmax","Claude Code",[2.45,2.89],true,"\u0001","\u0001"],["llm-speed-anatomy","speed-anatomy-total-by-size","1k prompt","Claude Haiku 4.5 · Claude Code","speed-anatomy-total-by-size/1k prompt","Total time per call by prompt size (1k prompt)",2.34,"seconds","2.34 s",3,"\u0001","minmax","Claude Code",[2.22,2.46],"\u0001","\u0001","\u0001"],["llm-speed-anatomy","speed-anatomy-total-by-size","16k prompt","Claude Haiku 4.5 · Claude Code","speed-anatomy-total-by-size/16k prompt","Total time per call by prompt size (16k prompt)",2.79,"seconds","2.79 s",3,"\u0001","minmax","Claude Code",[2.58,2.84],"\u0001","\u0001","\u0001"],["llm-speed-anatomy","speed-anatomy-total-by-size","64k prompt","Claude Haiku 4.5 · Claude Code","speed-anatomy-total-by-size/64k prompt","Total time per call by prompt size (64k prompt)",3.13,"seconds","3.13 s",3,"\u0001","minmax","Claude Code",[2.84,3.28],"\u0001","\u0001","\u0001"],["llm-speed-anatomy","speed-anatomy-lookup-correct","Exact answer","Claude Haiku 4.5 · Claude Code","speed-anatomy-lookup-correct","Exact lookup answers at the 1k, 16k and 64k prompt-size targets",1,"rate","100% (9/9)",9,[0.7009,1],"ci95","Claude Code","\u0001","\u0001","higher","\u0001"],["haiku-retry-or-escalate","retry-escalate-call-cost-by-task","Claude Haiku 4.5 · Claude Code","Interval merge fix","retry-escalate-call-cost-by-task/Interval merge fix","List-price cost of one call, Haiku 4.5 vs Sonnet 5.5, on each hard task (calculation): Interval merge fix",0.01789,"usd","$0.018",3,"\u0001","\u0001","Claude Code","\u0001",true,"\u0001","\u0001"],["haiku-retry-or-escalate","retry-escalate-call-cost-by-task","Claude Haiku 4.5 · Claude Code","DST day length","retry-escalate-call-cost-by-task/DST day length","List-price cost of one call, Haiku 4.5 vs Sonnet 5.5, on each hard task (calculation): DST day length",0.0293,"usd","$0.029",3,"\u0001","\u0001","Claude Code","\u0001",true,"\u0001","\u0001"],["haiku-retry-or-escalate","retry-escalate-call-cost-by-task","Claude Haiku 4.5 · Claude Code","CSV parser","retry-escalate-call-cost-by-task/CSV parser","List-price cost of one call, Haiku 4.5 vs Sonnet 5.5, on each hard task (calculation): CSV parser",0.02919,"usd","$0.029",3,"\u0001","\u0001","Claude Code","\u0001",true,"\u0001","\u0001"],["haiku-retry-or-escalate","retry-escalate-call-cost-by-task","Claude Haiku 4.5 · Claude Code","Event-loop order","retry-escalate-call-cost-by-task/Event-loop order","List-price cost of one call, Haiku 4.5 vs Sonnet 5.5, on each hard task (calculation): Event-loop order",0.03708,"usd","$0.037",3,"\u0001","\u0001","Claude Code","\u0001",true,"\u0001","\u0001"],["haiku-retry-or-escalate","retry-escalate-call-cost-by-task","Claude Haiku 4.5 · Claude Code","Room schedule","retry-escalate-call-cost-by-task/Room schedule","List-price cost of one call, Haiku 4.5 vs Sonnet 5.5, on each hard task (calculation): Room schedule",0.03578,"usd","$0.036",3,"\u0001","\u0001","Claude Code","\u0001",true,"\u0001","\u0001"],["haiku-retry-or-escalate","retry-escalate-call-cost-by-task","Claude Haiku 4.5 · Claude Code","SemVer regex","retry-escalate-call-cost-by-task/SemVer regex","List-price cost of one call, Haiku 4.5 vs Sonnet 5.5, on each hard task (calculation): SemVer regex",0.04217,"usd","$0.042",3,"\u0001","\u0001","Claude Code","\u0001",true,"\u0001","\u0001"],["haiku-retry-or-escalate","retry-escalate-call-cost-by-task","Claude Haiku 4.5 · Claude Code","Money refactor","retry-escalate-call-cost-by-task/Money refactor","List-price cost of one call, Haiku 4.5 vs Sonnet 5.5, on each hard task (calculation): Money refactor",0.02039,"usd","$0.020",3,"\u0001","\u0001","Claude Code","\u0001",true,"\u0001","\u0001"],["haiku-retry-or-escalate","retry-escalate-call-cost-by-task","Claude Haiku 4.5 · Claude Code","SQLite report query","retry-escalate-call-cost-by-task/SQLite report query","List-price cost of one call, Haiku 4.5 vs Sonnet 5.5, on each hard task (calculation): SQLite report query",0.03648,"usd","$0.036",3,"\u0001","\u0001","Claude Code","\u0001",true,"\u0001","\u0001"],["harder-tasks-head-to-head","harder-h2h-pass-rate","Strict pass","Claude Haiku 4.5 · Claude Code","harder-h2h-pass-rate/Strict pass","Pass rate on 4 harder tasks (Strict pass)",0,"rate","0% (0/12)",12,[0,0.2425],"ci95","Claude Code","\u0001","\u0001","higher","\u0001"],["harder-tasks-head-to-head","harder-h2h-pass-rate","Lenient (format misses counted)","Claude Haiku 4.5 · Claude Code","harder-h2h-pass-rate/Lenient (format misses counted)","Pass rate on 4 harder tasks (Lenient (format misses counted))",0,"rate","0% (0/12)",12,[0,0.2425],"ci95","Claude Code","\u0001","\u0001","higher","\u0001"],["harder-tasks-head-to-head","harder-h2h-tool-attempts","Tool attempt","Claude Haiku 4.5 · Claude Code","harder-h2h-tool-attempts","Calls that tried a tool although tools were off",0.0833,"rate","8% (1/12)",12,[0.0149,0.3539],"ci95","Claude Code","\u0001","\u0001","none","\u0001"],["harder-tasks-head-to-head","harder-h2h-pass-by-task","Claude Haiku 4.5 · Claude Code","10x10 nonogram","harder-h2h-pass-by-task/10x10 nonogram","Strict pass rate by task: 10x10 nonogram",0,"rate","0% (0/3)",3,[0,0.5615],"ci95","Claude Code","\u0001","\u0001","higher","\u0001"],["harder-tasks-head-to-head","harder-h2h-pass-by-task","Claude Haiku 4.5 · Claude Code","Sudoku, 22 givens","harder-h2h-pass-by-task/Sudoku, 22 givens","Strict pass rate by task: Sudoku, 22 givens",0,"rate","0% (0/3)",3,[0,0.5615],"ci95","Claude Code","\u0001","\u0001","higher","\u0001"],["harder-tasks-head-to-head","harder-h2h-pass-by-task","Claude Haiku 4.5 · Claude Code","6x6 Skyscrapers","harder-h2h-pass-by-task/6x6 Skyscrapers","Strict pass rate by task: 6x6 Skyscrapers",0,"rate","0% (0/3)",3,[0,0.5615],"ci95","Claude Code","\u0001","\u0001","higher","\u0001"],["harder-tasks-head-to-head","harder-h2h-pass-by-task","Claude Haiku 4.5 · Claude Code","Seeded shuffle output","harder-h2h-pass-by-task/Seeded shuffle output","Strict pass rate by task: Seeded shuffle output",0,"rate","0% (0/3)",3,[0,0.5615],"ci95","Claude Code","\u0001","\u0001","higher","\u0001"],["harder-tasks-head-to-head","harder-h2h-total-latency","Total time per call","Claude Haiku 4.5 · Claude Code","harder-h2h-total-latency","Total time per call on harder tasks",108.98,"seconds","109.0 s",10,"\u0001","minmax","Claude Code",[25.73,223.95],"\u0001","\u0001","\u0001"],["harder-tasks-head-to-head","harder-h2h-output-tokens","Output tokens","Claude Haiku 4.5 · Claude Code","harder-h2h-output-tokens/Output tokens","Output tokens per call on harder tasks (Output tokens)",12508,"tokens","12,508",10,"\u0001","minmax","Claude Code",[2965,26532],"\u0001","none","\u0001"]]}],["claude-fable-5-1","Claude Fable 5.1","Anthropic","model","Anthropic’s highest-priced Claude model in these studies, measured through Claude Code.",["Claude Fable 5.1","Fable 5.1"],{"$k":["studySlug","chartId","series","point","metric","label","value","unit","display","n","ci","spanKind","context","range","calculation","polarity"],"$r":[["model-head-to-head","h2h-pass-rate","Pass rate","Claude Fable 5.1 · Claude Code","h2h-pass-rate","Pass rate on five validated tasks",1,"rate","100% (15/15)",15,[0.7961,1],"ci95","Claude Code · five short validated tasks","\u0001","\u0001","\u0001"],["model-head-to-head","h2h-total-latency","Total time per call","Claude Fable 5.1 · Claude Code","h2h-total-latency","Total time per call",1.94,"seconds","1.94 s",15,"\u0001","minmax","Claude Code · five short validated tasks",[1.41,9.83],"\u0001","\u0001"],["model-head-to-head","h2h-first-useful-latency","Time to first useful output","Claude Fable 5.1 · Claude Code","h2h-first-useful-latency","Time to first useful output",1.2,"seconds","1.20 s",15,"\u0001","minmax","Claude Code · five short validated tasks",[0.95,7.9],"\u0001","\u0001"],["model-head-to-head","h2h-input-tokens","Cache read","Claude Fable 5.1 · Claude Code","h2h-input-tokens/Cache read","Input tokens per call: what the CLI sends (Cache read)",2760,"tokens","2,760",15,"\u0001","\u0001","Claude Code · five short validated tasks","\u0001","\u0001","\u0001"],["model-head-to-head","h2h-input-tokens","Other input","Claude Fable 5.1 · Claude Code","h2h-input-tokens/Other input","Input tokens per call: what the CLI sends (Other input)",473,"tokens","473",15,"\u0001","\u0001","Claude Code · five short validated tasks","\u0001","\u0001","\u0001"],["model-head-to-head","h2h-output-tokens","Output tokens","Claude Fable 5.1 · Claude Code","h2h-output-tokens/Output tokens","Output tokens per call (Output tokens)",64,"tokens","64",15,"\u0001","\u0001","Claude Code · five short validated tasks","\u0001","\u0001","\u0001"],["model-head-to-head","h2h-list-price-per-call","Cost per call","Claude Fable 5.1 · Claude Code","h2h-list-price-per-call","List-price cost per call (calculation)",0.00987,"usd","$0.0099",15,"\u0001","minmax","Claude Code · five short validated tasks",[0.0049,0.05843],true,"\u0001"],["model-head-to-head","h2h-cost-per-pass","Cost per pass","Claude Fable 5.1 · Claude Code","h2h-cost-per-pass","List-price cost per passing answer (calculation)",0.02054,"usd","$0.021",15,"\u0001","\u0001","Claude Code · five short validated tasks","\u0001",true,"\u0001"],["hard-model-head-to-head","hard-h2h-pass-rate","Strict pass","Claude Fable 5.1 · Claude Code","hard-h2h-pass-rate/Strict pass","Pass rate on eight hard tasks (Strict pass)",1,"rate","100% (24/24)",24,[0.862,1],"ci95","Claude Code · eight hard validated tasks","\u0001","\u0001","\u0001"],["hard-model-head-to-head","hard-h2h-pass-rate","Lenient (format misses counted)","Claude Fable 5.1 · Claude Code","hard-h2h-pass-rate/Lenient (format misses counted)","Pass rate on eight hard tasks (Lenient (format misses counted))",1,"rate","100% (24/24)",24,[0.862,1],"ci95","Claude Code · eight hard validated tasks","\u0001","\u0001","\u0001"],["hard-model-head-to-head","hard-h2h-total-latency","Total time per call on hard tasks (separate batches)","Claude Fable 5.1 · Claude Code","hard-h2h-total-latency","Total time per call on hard tasks (separate batches)",16.13,"seconds","16.1 s",24,"\u0001","minmax","Claude Code · eight hard validated tasks",[4.46,90],"\u0001","\u0001"],["hard-model-head-to-head","hard-h2h-first-useful-latency","Time to first useful output on hard tasks","Claude Fable 5.1 · Claude Code","hard-h2h-first-useful-latency","Time to first useful output on hard tasks",11.63,"seconds","11.6 s",24,"\u0001","minmax","Claude Code · eight hard validated tasks",[2,85.33],"\u0001","\u0001"],["hard-model-head-to-head","hard-h2h-output-tokens","Output tokens","Claude Fable 5.1 · Claude Code","hard-h2h-output-tokens/Output tokens","Output tokens per call on hard tasks (Output tokens)",1366,"tokens","1,366",24,"\u0001","\u0001","Claude Code · eight hard validated tasks","\u0001","\u0001","\u0001"],["hard-model-head-to-head","hard-h2h-cost-per-pass","Cost per strict pass","Claude Fable 5.1 · Claude Code","hard-h2h-cost-per-pass","List-price cost per strict pass on hard tasks (calculation)",0.09331,"usd","$0.093",24,"\u0001","\u0001","Claude Code · eight hard validated tasks","\u0001",true,"\u0001"],["cost-thought-experiments","repriced-cost-per-resolved","Repriced cost per resolved instance","Claude Fable 5.1","repriced-cost-per-resolved","Thought experiment: the same tokens at other list prices",12.852,"usd","$12.85","\u0001","\u0001","\u0001","calculation: Agent’s recorded tokens at this model’s list price","\u0001",true,"\u0001"],["cost-thought-experiments","prompt-cache-savings","With caching (as recorded)","Claude Fable 5.1","prompt-cache-savings/With caching (as recorded)","Thought experiment: what prompt caching saved (With caching (as recorded))",321.31,"usd","$321.31","\u0001","\u0001","\u0001","calculation: Agent’s recorded tokens at this model’s list price","\u0001",true,"\u0001"],["cost-thought-experiments","prompt-cache-savings","Without caching","Claude Fable 5.1","prompt-cache-savings/Without caching","Thought experiment: what prompt caching saved (Without caching)",1716.65,"usd","$1716.65","\u0001","\u0001","\u0001","calculation: Agent’s recorded tokens at this model’s list price","\u0001",true,"\u0001"],["prompt-cache-break-even","cache-break-even-reads","1-hour write (2× input), whole prefix new","Claude Fable 5.1","cache-break-even-reads/1-hour write (2× input), whole prefix new","Reuses before a cached prefix costs less, by model and write type (calculation) (1-hour write (2× input), whole prefix new)",1.03,"score","1.03","\u0001","\u0001","\u0001","calculation: Agent’s recorded tokens at this model’s list price","\u0001",true,"\u0001"],["prompt-cache-break-even","cache-break-even-reads","1-hour write, pooled n = 6 session share, 19% already cached (as recorded)","Claude Fable 5.1","cache-break-even-reads/1-hour write, pooled n = 6 session share, 19% already cached (as recorded)","Reuses before a cached prefix costs less, by model and write type (calculation) (1-hour write, pooled n = 6 session share, 19% already cached (as recorded))",0.65,"score","0.65","\u0001","\u0001","\u0001","calculation: Agent’s recorded tokens at this model’s list price","\u0001",true,"\u0001"],["prompt-cache-break-even","cache-break-even-reads","5-minute write (1.25× input, an assumption)","Claude Fable 5.1","cache-break-even-reads/5-minute write (1.25× input, an assumption)","Reuses before a cached prefix costs less, by model and write type (calculation) (5-minute write (1.25× input, an assumption))",0.26,"score","0.26","\u0001","\u0001","\u0001","calculation: Agent’s recorded tokens at this model’s list price","\u0001",true,"\u0001"],["thinking-token-bill","thinking-bill-share","Median call","Claude Fable 5.1 · Claude Code","thinking-bill-share","Reasoning share of output tokens per call on hard tasks (calculation)",64.24,"percent","64.2%",24,"\u0001","minmax","Claude Code",[23.44,97.19],true,"none"],["thinking-token-bill","thinking-bill-cost-per-call","Reasoning (output tokens)","Claude Fable 5.1 · Claude Code","thinking-bill-cost-per-call/Reasoning (output tokens)","List-price cost per call: reasoning, remaining output and input (calculation) (Reasoning (output tokens))",0.053696,"usd","$0.054",24,"\u0001","\u0001","Claude Code","\u0001",true,"none"],["thinking-token-bill","thinking-bill-cost-per-call","Remaining output (visible-answer estimate)","Claude Fable 5.1 · Claude Code","thinking-bill-cost-per-call/Remaining output (visible-answer estimate)","List-price cost per call: reasoning, remaining output and input (calculation) (Remaining output (visible-answer estimate))",0.018702,"usd","$0.019",24,"\u0001","\u0001","Claude Code","\u0001",true,"none"],["thinking-token-bill","thinking-bill-cost-per-call","Input (prompt, cache priced)","Claude Fable 5.1 · Claude Code","thinking-bill-cost-per-call/Input (prompt, cache priced)","List-price cost per call: reasoning, remaining output and input (calculation) (Input (prompt, cache priced))",0.02091,"usd","$0.021",24,"\u0001","\u0001","Claude Code","\u0001",true,"none"],["thinking-token-bill","thinking-bill-short-vs-hard","Eight hard tasks","Claude Fable 5.1 · Claude Code","thinking-bill-short-vs-hard/Eight hard tasks","Reasoning share on short tasks vs hard tasks (calculation) (Eight hard tasks)",64.24,"percent","64.2%",24,"\u0001","minmax","Claude Code",[23.44,97.19],true,"none"],["thinking-token-bill","thinking-bill-short-vs-hard","Five short tasks","Claude Fable 5.1 · Claude Code","thinking-bill-short-vs-hard/Five short tasks","Reasoning share on short tasks vs hard tasks (calculation) (Five short tasks)",0,"percent","0%",15,"\u0001","minmax","Claude Code",[0,74.01],true,"none"],["llm-speed-anatomy","speed-anatomy-first-text","Time to first text","Claude Fable 5.1 · Claude Code","speed-anatomy-first-text","Time to first text: a 250-line answer, six models",4.43,"seconds","4.43 s",4,"\u0001","minmax","Claude Code",[2.27,4.64],"\u0001","\u0001"],["llm-speed-anatomy","speed-anatomy-output-speed","Visible tokens per second","Claude Fable 5.1 · Claude Code","speed-anatomy-output-speed","Output speed after the first text: visible tokens per second (calculation)",122.6,"tokens","123",4,"\u0001","minmax","Claude Code",[120.9,131.4],true,"\u0001"],["llm-speed-anatomy","speed-anatomy-chars-per-second","Characters per second","Claude Fable 5.1 · Claude Code","speed-anatomy-chars-per-second","Output speed in characters per second after the first text (calculation)",273,"count","273",4,"\u0001","minmax","Claude Code",[270,293],true,"\u0001"]]}],["gpt-6-1-sol-codex-cli","GPT-6.1 Sol (Codex CLI)","OpenAI","model","OpenAI’s GPT-6.1 Sol model run through the Codex CLI at low, medium and high effort.",["GPT-6.1 Sol · Codex CLI"],{"$k":["studySlug","chartId","series","point","metric","label","value","unit","display","n","ci","spanKind","context","range","calculation","polarity"],"$r":[["model-head-to-head","h2h-pass-rate","Pass rate","GPT-6.1 Sol (high) · Codex CLI","h2h-pass-rate","Pass rate on five validated tasks",1,"rate","100% (15/15)",15,[0.7961,1],"ci95","Codex CLI · effort high · five short validated tasks","\u0001","\u0001","\u0001"],["model-head-to-head","h2h-pass-rate","Pass rate","GPT-6.1 Sol (medium) · Codex CLI","h2h-pass-rate","Pass rate on five validated tasks",1,"rate","100% (15/15)",15,[0.7961,1],"ci95","Codex CLI · effort medium · five short validated tasks","\u0001","\u0001","\u0001"],["model-head-to-head","h2h-pass-rate","Pass rate","GPT-6.1 Sol (low) · Codex CLI","h2h-pass-rate","Pass rate on five validated tasks",1,"rate","100% (10/10)",10,[0.7225,1],"ci95","Codex CLI · effort low · five short validated tasks","\u0001","\u0001","\u0001"],["model-head-to-head","h2h-total-latency","Total time per call","GPT-6.1 Sol (high) · Codex CLI","h2h-total-latency","Total time per call",5.6,"seconds","5.60 s",15,"\u0001","minmax","Codex CLI · effort high · five short validated tasks",[4.05,19.52],"\u0001","\u0001"],["model-head-to-head","h2h-total-latency","Total time per call","GPT-6.1 Sol (medium) · Codex CLI","h2h-total-latency","Total time per call",5.65,"seconds","5.65 s",15,"\u0001","minmax","Codex CLI · effort medium · five short validated tasks",[4.1,25.46],"\u0001","\u0001"],["model-head-to-head","h2h-total-latency","Total time per call","GPT-6.1 Sol (low) · Codex CLI","h2h-total-latency","Total time per call",6.26,"seconds","6.26 s",10,"\u0001","minmax","Codex CLI · effort low · five short validated tasks",[4.65,10.47],"\u0001","\u0001"],["model-head-to-head","h2h-first-useful-latency","Time to first useful output","GPT-6.1 Sol (high) · Codex CLI","h2h-first-useful-latency","Time to first useful output",5.32,"seconds","5.32 s",15,"\u0001","minmax","Codex CLI · effort high · five short validated tasks",[3.64,16.37],"\u0001","\u0001"],["model-head-to-head","h2h-first-useful-latency","Time to first useful output","GPT-6.1 Sol (medium) · Codex CLI","h2h-first-useful-latency","Time to first useful output",5.05,"seconds","5.05 s",15,"\u0001","minmax","Codex CLI · effort medium · five short validated tasks",[3.36,17.82],"\u0001","\u0001"],["model-head-to-head","h2h-first-useful-latency","Time to first useful output","GPT-6.1 Sol (low) · Codex CLI","h2h-first-useful-latency","Time to first useful output",5.14,"seconds","5.14 s",10,"\u0001","minmax","Codex CLI · effort low · five short validated tasks",[4.02,8.5],"\u0001","\u0001"],["model-head-to-head","h2h-input-tokens","Cache read","GPT-6.1 Sol (high) · Codex CLI","h2h-input-tokens/Cache read","Input tokens per call: what the CLI sends (Cache read)",6716,"tokens","6,716",15,"\u0001","\u0001","Codex CLI · effort high · five short validated tasks","\u0001","\u0001","\u0001"],["model-head-to-head","h2h-input-tokens","Cache read","GPT-6.1 Sol (medium) · Codex CLI","h2h-input-tokens/Cache read","Input tokens per call: what the CLI sends (Cache read)",5180,"tokens","5,180",15,"\u0001","\u0001","Codex CLI · effort medium · five short validated tasks","\u0001","\u0001","\u0001"],["model-head-to-head","h2h-input-tokens","Cache read","GPT-6.1 Sol (low) · Codex CLI","h2h-input-tokens/Cache read","Input tokens per call: what the CLI sends (Cache read)",8064,"tokens","8,064",10,"\u0001","\u0001","Codex CLI · effort low · five short validated tasks","\u0001","\u0001","\u0001"],["model-head-to-head","h2h-input-tokens","Other input","GPT-6.1 Sol (high) · Codex CLI","h2h-input-tokens/Other input","Input tokens per call: what the CLI sends (Other input)",5406,"tokens","5,406",15,"\u0001","\u0001","Codex CLI · effort high · five short validated tasks","\u0001","\u0001","\u0001"],["model-head-to-head","h2h-input-tokens","Other input","GPT-6.1 Sol (medium) · Codex CLI","h2h-input-tokens/Other input","Input tokens per call: what the CLI sends (Other input)",6943,"tokens","6,943",15,"\u0001","\u0001","Codex CLI · effort medium · five short validated tasks","\u0001","\u0001","\u0001"],["model-head-to-head","h2h-input-tokens","Other input","GPT-6.1 Sol (low) · Codex CLI","h2h-input-tokens/Other input","Input tokens per call: what the CLI sends (Other input)",4059,"tokens","4,059",10,"\u0001","\u0001","Codex CLI · effort low · five short validated tasks","\u0001","\u0001","\u0001"],["model-head-to-head","h2h-output-tokens","Output tokens","GPT-6.1 Sol (high) · Codex CLI","h2h-output-tokens/Output tokens","Output tokens per call (Output tokens)",42,"tokens","42",15,"\u0001","\u0001","Codex CLI · effort high · five short validated tasks","\u0001","\u0001","\u0001"],["model-head-to-head","h2h-output-tokens","Output tokens","GPT-6.1 Sol (medium) · Codex CLI","h2h-output-tokens/Output tokens","Output tokens per call (Output tokens)",42,"tokens","42",15,"\u0001","\u0001","Codex CLI · effort medium · five short validated tasks","\u0001","\u0001","\u0001"],["model-head-to-head","h2h-output-tokens","Output tokens","GPT-6.1 Sol (low) · Codex CLI","h2h-output-tokens/Output tokens","Output tokens per call (Output tokens)",42,"tokens","42",10,"\u0001","\u0001","Codex CLI · effort low · five short validated tasks","\u0001","\u0001","\u0001"],["model-head-to-head","h2h-list-price-per-call","Cost per call","GPT-6.1 Sol (high) · Codex CLI","h2h-list-price-per-call","List-price cost per call (calculation)",0.01047,"usd","$0.010",15,"\u0001","minmax","Codex CLI · effort high · five short validated tasks",[0.0066,0.02812],true,"\u0001"],["model-head-to-head","h2h-list-price-per-call","Cost per call","GPT-6.1 Sol (medium) · Codex CLI","h2h-list-price-per-call","List-price cost per call (calculation)",0.01018,"usd","$0.010",15,"\u0001","minmax","Codex CLI · effort medium · five short validated tasks",[0.0054,0.02686],true,"\u0001"],["model-head-to-head","h2h-list-price-per-call","Cost per call","GPT-6.1 Sol (low) · Codex CLI","h2h-list-price-per-call","List-price cost per call (calculation)",0.00769,"usd","$0.0077",10,"\u0001","minmax","Codex CLI · effort low · five short validated tasks",[0.00742,0.02649],true,"\u0001"],["model-head-to-head","h2h-cost-per-pass","Cost per pass","GPT-6.1 Sol (low) · Codex CLI","h2h-cost-per-pass","List-price cost per passing answer (calculation)",0.00998,"usd","$0.010",10,"\u0001","\u0001","Codex CLI · effort low · five short validated tasks","\u0001",true,"\u0001"],["model-head-to-head","h2h-cost-per-pass","Cost per pass","GPT-6.1 Sol (high) · Codex CLI","h2h-cost-per-pass","List-price cost per passing answer (calculation)",0.01322,"usd","$0.013",15,"\u0001","\u0001","Codex CLI · effort high · five short validated tasks","\u0001",true,"\u0001"],["model-head-to-head","h2h-cost-per-pass","Cost per pass","GPT-6.1 Sol (medium) · Codex CLI","h2h-cost-per-pass","List-price cost per passing answer (calculation)",0.01564,"usd","$0.016",15,"\u0001","\u0001","Codex CLI · effort medium · five short validated tasks","\u0001",true,"\u0001"],["hard-model-head-to-head","hard-h2h-pass-rate","Strict pass","GPT-6.1 Sol (medium) · Codex CLI","hard-h2h-pass-rate/Strict pass","Pass rate on eight hard tasks (Strict pass)",1,"rate","100% (16/16)",16,[0.8064,1],"ci95","Codex CLI · effort medium · eight hard validated tasks","\u0001","\u0001","\u0001"],["hard-model-head-to-head","hard-h2h-pass-rate","Strict pass","GPT-6.1 Sol (high) · Codex CLI","hard-h2h-pass-rate/Strict pass","Pass rate on eight hard tasks (Strict pass)",1,"rate","100% (16/16)",16,[0.8064,1],"ci95","Codex CLI · effort high · eight hard validated tasks","\u0001","\u0001","\u0001"],["hard-model-head-to-head","hard-h2h-pass-rate","Lenient (format misses counted)","GPT-6.1 Sol (medium) · Codex CLI","hard-h2h-pass-rate/Lenient (format misses counted)","Pass rate on eight hard tasks (Lenient (format misses counted))",1,"rate","100% (16/16)",16,[0.8064,1],"ci95","Codex CLI · effort medium · eight hard validated tasks","\u0001","\u0001","\u0001"],["hard-model-head-to-head","hard-h2h-pass-rate","Lenient (format misses counted)","GPT-6.1 Sol (high) · Codex CLI","hard-h2h-pass-rate/Lenient (format misses counted)","Pass rate on eight hard tasks (Lenient (format misses counted))",1,"rate","100% (16/16)",16,[0.8064,1],"ci95","Codex CLI · effort high · eight hard validated tasks","\u0001","\u0001","\u0001"],["hard-model-head-to-head","hard-h2h-total-latency","Total time per call on hard tasks (separate batches)","GPT-6.1 Sol (medium) · Codex CLI","hard-h2h-total-latency","Total time per call on hard tasks (separate batches)",13.11,"seconds","13.1 s",16,"\u0001","minmax","Codex CLI · effort medium · eight hard validated tasks",[8.54,61.6],"\u0001","\u0001"],["hard-model-head-to-head","hard-h2h-total-latency","Total time per call on hard tasks (separate batches)","GPT-6.1 Sol (high) · Codex CLI","hard-h2h-total-latency","Total time per call on hard tasks (separate batches)",18.12,"seconds","18.1 s",16,"\u0001","minmax","Codex CLI · effort high · eight hard validated tasks",[11.67,92.21],"\u0001","\u0001"],["hard-model-head-to-head","hard-h2h-first-useful-latency","Time to first useful output on hard tasks","GPT-6.1 Sol (medium) · Codex CLI","hard-h2h-first-useful-latency","Time to first useful output on hard tasks",10.23,"seconds","10.2 s",16,"\u0001","minmax","Codex CLI · effort medium · eight hard validated tasks",[6.09,40.41],"\u0001","\u0001"],["hard-model-head-to-head","hard-h2h-first-useful-latency","Time to first useful output on hard tasks","GPT-6.1 Sol (high) · Codex CLI","hard-h2h-first-useful-latency","Time to first useful output on hard tasks",12.69,"seconds","12.7 s",16,"\u0001","minmax","Codex CLI · effort high · eight hard validated tasks",[8.93,75.91],"\u0001","\u0001"],["hard-model-head-to-head","hard-h2h-output-tokens","Output tokens","GPT-6.1 Sol (medium) · Codex CLI","hard-h2h-output-tokens/Output tokens","Output tokens per call on hard tasks (Output tokens)",335,"tokens","335",16,"\u0001","\u0001","Codex CLI · effort medium · eight hard validated tasks","\u0001","\u0001","\u0001"],["hard-model-head-to-head","hard-h2h-output-tokens","Output tokens","GPT-6.1 Sol (high) · Codex CLI","hard-h2h-output-tokens/Output tokens","Output tokens per call on hard tasks (Output tokens)",436,"tokens","436",16,"\u0001","\u0001","Codex CLI · effort high · eight hard validated tasks","\u0001","\u0001","\u0001"],["hard-model-head-to-head","hard-h2h-cost-per-pass","Cost per strict pass","GPT-6.1 Sol (high) · Codex CLI","hard-h2h-cost-per-pass","List-price cost per strict pass on hard tasks (calculation)",0.01514,"usd","$0.015",16,"\u0001","\u0001","Codex CLI · effort high · eight hard validated tasks","\u0001",true,"\u0001"],["hard-model-head-to-head","hard-h2h-cost-per-pass","Cost per strict pass","GPT-6.1 Sol (medium) · Codex CLI","hard-h2h-cost-per-pass","List-price cost per strict pass on hard tasks (calculation)",0.02564,"usd","$0.026",16,"\u0001","\u0001","Codex CLI · effort medium · eight hard validated tasks","\u0001",true,"\u0001"],["coding-agents-head-to-head","coding-agents-pass-rate","Passed every hidden check","GPT-6.1 Sol (medium, tester’s AGENTS.md) · Codex CLI","coding-agents-pass-rate","Coding sessions that passed every hidden check",1,"rate","100% (12/12)",12,[0.7575,1],"ci95","Codex CLI · effort medium · tester’s AGENTS.md · six small repository tasks with hidden tests","\u0001","\u0001","\u0001"],["coding-agents-head-to-head","coding-agents-wall-time","Wall time per session","GPT-6.1 Sol (medium, tester’s AGENTS.md) · Codex CLI","coding-agents-wall-time","Time per coding session",113.4,"seconds","113.4 s",12,"\u0001","minmax","Codex CLI · effort medium · tester’s AGENTS.md · six small repository tasks with hidden tests",[78.5,221.9],"\u0001","\u0001"],["coding-agents-head-to-head","coding-agents-tool-calls","Tool calls per session","GPT-6.1 Sol (medium, tester’s AGENTS.md) · Codex CLI","coding-agents-tool-calls","Tool calls per coding session",12.5,"calls","12.5",12,"\u0001","minmax","Codex CLI · effort medium · tester’s AGENTS.md · six small repository tasks with hidden tests",[8,18],"\u0001","\u0001"],["coding-agents-head-to-head","coding-agents-cost-per-pass","List-price cost per pass","GPT-6.1 Sol (medium, tester’s AGENTS.md) · Codex CLI","coding-agents-cost-per-pass","List-price cost per passing coding session (calculation)",0.0978,"usd","$0.098",12,"\u0001","\u0001","Codex CLI · effort medium · tester’s AGENTS.md · six small repository tasks with hidden tests","\u0001",true,"\u0001"],["effort-ladder","effort-ladder-pass-rate","Strict pass","GPT-6.1 Sol (low) · Codex CLI","effort-ladder-pass-rate","Strict pass rate by effort on eight hard tasks",1,"rate","100% (16/16)",16,[0.8064,1],"ci95","Codex CLI · effort low · eight hard validated tasks, effort ladder","\u0001","\u0001","\u0001"],["effort-ladder","effort-ladder-pass-rate","Strict pass","GPT-6.1 Sol (medium) · Codex CLI","effort-ladder-pass-rate","Strict pass rate by effort on eight hard tasks",1,"rate","100% (16/16)",16,[0.8064,1],"ci95","Codex CLI · effort medium · eight hard validated tasks, effort ladder","\u0001","\u0001","\u0001"],["effort-ladder","effort-ladder-pass-rate","Strict pass","GPT-6.1 Sol (high) · Codex CLI","effort-ladder-pass-rate","Strict pass rate by effort on eight hard tasks",1,"rate","100% (16/16)",16,[0.8064,1],"ci95","Codex CLI · effort high · eight hard validated tasks, effort ladder","\u0001","\u0001","\u0001"],["effort-ladder","effort-ladder-total-latency","Total time per call by effort on hard tasks","GPT-6.1 Sol (low) · Codex CLI","effort-ladder-total-latency","Total time per call by effort on hard tasks",13.62,"seconds","13.6 s",16,"\u0001","minmax","Codex CLI · effort low · eight hard validated tasks, effort ladder",[7.94,44.29],"\u0001","\u0001"],["effort-ladder","effort-ladder-total-latency","Total time per call by effort on hard tasks","GPT-6.1 Sol (medium) · Codex CLI","effort-ladder-total-latency","Total time per call by effort on hard tasks",13.11,"seconds","13.1 s",16,"\u0001","minmax","Codex CLI · effort medium · eight hard validated tasks, effort ladder",[8.54,61.6],"\u0001","\u0001"],["effort-ladder","effort-ladder-total-latency","Total time per call by effort on hard tasks","GPT-6.1 Sol (high) · Codex CLI","effort-ladder-total-latency","Total time per call by effort on hard tasks",18.12,"seconds","18.1 s",16,"\u0001","minmax","Codex CLI · effort high · eight hard validated tasks, effort ladder",[11.67,92.21],"\u0001","\u0001"],["effort-ladder","effort-ladder-output-tokens","Output tokens","GPT-6.1 Sol (low) · Codex CLI","effort-ladder-output-tokens/Output tokens","Output tokens per call by effort on hard tasks (Output tokens)",284,"tokens","284",16,"\u0001","\u0001","Codex CLI · effort low · eight hard validated tasks, effort ladder","\u0001","\u0001","\u0001"],["effort-ladder","effort-ladder-output-tokens","Output tokens","GPT-6.1 Sol (medium) · Codex CLI","effort-ladder-output-tokens/Output tokens","Output tokens per call by effort on hard tasks (Output tokens)",335,"tokens","335",16,"\u0001","\u0001","Codex CLI · effort medium · eight hard validated tasks, effort ladder","\u0001","\u0001","\u0001"],["effort-ladder","effort-ladder-output-tokens","Output tokens","GPT-6.1 Sol (high) · Codex CLI","effort-ladder-output-tokens/Output tokens","Output tokens per call by effort on hard tasks (Output tokens)",436,"tokens","436",16,"\u0001","\u0001","Codex CLI · effort high · eight hard validated tasks, effort ladder","\u0001","\u0001","\u0001"],["effort-ladder","effort-ladder-cost-per-pass","Cost per strict pass","GPT-6.1 Sol (low) · Codex CLI","effort-ladder-cost-per-pass","List-price cost per strict pass by effort (calculation)",0.01284,"usd","$0.013",16,"\u0001","\u0001","Codex CLI · effort low · eight hard validated tasks, effort ladder","\u0001",true,"\u0001"],["effort-ladder","effort-ladder-cost-per-pass","Cost per strict pass","GPT-6.1 Sol (medium) · Codex CLI","effort-ladder-cost-per-pass","List-price cost per strict pass by effort (calculation)",0.02564,"usd","$0.026",16,"\u0001","\u0001","Codex CLI · effort medium · eight hard validated tasks, effort ladder","\u0001",true,"\u0001"],["effort-ladder","effort-ladder-cost-per-pass","Cost per strict pass","GPT-6.1 Sol (high) · Codex CLI","effort-ladder-cost-per-pass","List-price cost per strict pass by effort (calculation)",0.01514,"usd","$0.015",16,"\u0001","\u0001","Codex CLI · effort high · eight hard validated tasks, effort ladder","\u0001",true,"\u0001"],["caching-consistency","consistency-pass-rate","Exact number","GPT-6.1 Sol (medium) · Codex CLI","consistency-pass-rate/Exact number","Same prompt, 10 times: strict pass rate (Exact number)",1,"rate","100% (10/10)",10,[0.7225,1],"ci95","Codex CLI · effort medium · same prompt repeated 10 times","\u0001","\u0001","\u0001"],["caching-consistency","consistency-pass-rate","JSON object","GPT-6.1 Sol (medium) · Codex CLI","consistency-pass-rate/JSON object","Same prompt, 10 times: strict pass rate (JSON object)",1,"rate","100% (10/10)",10,[0.7225,1],"ci95","Codex CLI · effort medium · same prompt repeated 10 times","\u0001","\u0001","\u0001"],["caching-consistency","consistency-pass-rate","Code fix","GPT-6.1 Sol (medium) · Codex CLI","consistency-pass-rate/Code fix","Same prompt, 10 times: strict pass rate (Code fix)",1,"rate","100% (10/10)",10,[0.7225,1],"ci95","Codex CLI · effort medium · same prompt repeated 10 times","\u0001","\u0001","\u0001"],["caching-consistency","consistency-distinct-answers","Exact number","GPT-6.1 Sol (medium) · Codex CLI","consistency-distinct-answers/Exact number","Same prompt, 10 times: how many different answers (Exact number)",1,"count","1",10,"\u0001","\u0001","Codex CLI · effort medium · same prompt repeated 10 times","\u0001","\u0001","\u0001"],["caching-consistency","consistency-distinct-answers","JSON object","GPT-6.1 Sol (medium) · Codex CLI","consistency-distinct-answers/JSON object","Same prompt, 10 times: how many different answers (JSON object)",1,"count","1",10,"\u0001","\u0001","Codex CLI · effort medium · same prompt repeated 10 times","\u0001","\u0001","\u0001"],["caching-consistency","consistency-distinct-answers","Code fix","GPT-6.1 Sol (medium) · Codex CLI","consistency-distinct-answers/Code fix","Same prompt, 10 times: how many different answers (Code fix)",6,"count","6",10,"\u0001","\u0001","Codex CLI · effort medium · same prompt repeated 10 times","\u0001","\u0001","\u0001"],["caching-consistency","consistency-latency-spread","Exact number","GPT-6.1 Sol (medium) · Codex CLI","consistency-latency-spread/Exact number","Same prompt, 10 times: time per call (Exact number)",13.38,"seconds","13.4 s",10,"\u0001","minmax","Codex CLI · effort medium · same prompt repeated 10 times",[12.29,17.97],"\u0001","\u0001"],["caching-consistency","consistency-latency-spread","JSON object","GPT-6.1 Sol (medium) · Codex CLI","consistency-latency-spread/JSON object","Same prompt, 10 times: time per call (JSON object)",6.42,"seconds","6.42 s",10,"\u0001","minmax","Codex CLI · effort medium · same prompt repeated 10 times",[5.25,8.26],"\u0001","\u0001"],["caching-consistency","consistency-latency-spread","Code fix","GPT-6.1 Sol (medium) · Codex CLI","consistency-latency-spread/Code fix","Same prompt, 10 times: time per call (Code fix)",11.29,"seconds","11.3 s",10,"\u0001","minmax","Codex CLI · effort medium · same prompt repeated 10 times",[9.08,14.85],"\u0001","\u0001"],["cli-model-latency-tokens","cli-vs-api-exact-reply-latency","Total time","Codex CLI · GPT-6.1 Sol · low","cli-vs-api-exact-reply-latency/Total time","CLI vs API: time for a one-line answer (Total time)",4.18,"seconds","4.18 s",5,"\u0001","minmax","Codex CLI · effort low · fixed exact reply, 5 runs",[3.86,4.53],"\u0001","\u0001"],["cli-model-latency-tokens","cli-vs-api-exact-reply-latency","Total time","Codex CLI · GPT-6.1 Sol · high","cli-vs-api-exact-reply-latency/Total time","CLI vs API: time for a one-line answer (Total time)",4.19,"seconds","4.19 s",5,"\u0001","minmax","Codex CLI · effort high · fixed exact reply, 5 runs",[3.81,4.69],"\u0001","\u0001"],["cli-model-latency-tokens","cli-vs-api-exact-reply-latency","First useful output","Codex CLI · GPT-6.1 Sol · low","cli-vs-api-exact-reply-latency/First useful output","CLI vs API: time for a one-line answer (First useful output)",3.75,"seconds","3.75 s",5,"\u0001","minmax","Codex CLI · effort low · fixed exact reply, 5 runs",[3.44,4.1],"\u0001","\u0001"],["cli-model-latency-tokens","cli-vs-api-exact-reply-latency","First useful output","Codex CLI · GPT-6.1 Sol · high","cli-vs-api-exact-reply-latency/First useful output","CLI vs API: time for a one-line answer (First useful output)",3.79,"seconds","3.79 s",5,"\u0001","minmax","Codex CLI · effort high · fixed exact reply, 5 runs",[3.37,4.3],"\u0001","\u0001"],["cli-model-latency-tokens","cli-vs-api-small-coding-latency","Total time","Codex CLI · GPT-6.1 Sol · low","cli-vs-api-small-coding-latency/Total time","CLI vs API: time for a small coding task (Total time)",14.15,"seconds","14.2 s",3,"\u0001","minmax","Codex CLI · effort low · small coding task, 3 runs",[13.02,14.41],"\u0001","\u0001"],["cli-model-latency-tokens","cli-vs-api-small-coding-latency","Total time","Codex CLI · GPT-6.1 Sol · high","cli-vs-api-small-coding-latency/Total time","CLI vs API: time for a small coding task (Total time)",17.85,"seconds","17.9 s",3,"\u0001","minmax","Codex CLI · effort high · small coding task, 3 runs",[17.68,22.42],"\u0001","\u0001"],["cli-model-latency-tokens","cli-vs-api-small-coding-latency","First useful output","Codex CLI · GPT-6.1 Sol · low","cli-vs-api-small-coding-latency/First useful output","CLI vs API: time for a small coding task (First useful output)",13.6,"seconds","13.6 s",3,"\u0001","minmax","Codex CLI · effort low · small coding task, 3 runs",[12.52,13.83],"\u0001","\u0001"],["cli-model-latency-tokens","cli-vs-api-small-coding-latency","First useful output","Codex CLI · GPT-6.1 Sol · high","cli-vs-api-small-coding-latency/First useful output","CLI vs API: time for a small coding task (First useful output)",17.27,"seconds","17.3 s",3,"\u0001","minmax","Codex CLI · effort high · small coding task, 3 runs",[17.13,21.86],"\u0001","\u0001"],["cli-model-latency-tokens","cli-vs-api-prompt-overhead","Input tokens","Codex CLI · GPT-6.1 Sol · low","cli-vs-api-prompt-overhead","Hidden prompt: input tokens for the same one-line request",19551,"tokens","19,551",5,"\u0001","\u0001","Codex CLI · effort low · short fixed tasks","\u0001","\u0001","\u0001"],["cli-model-latency-tokens","cli-vs-api-prompt-overhead","Input tokens","Codex CLI · GPT-6.1 Sol · high","cli-vs-api-prompt-overhead","Hidden prompt: input tokens for the same one-line request",19555,"tokens","19,555",5,"\u0001","\u0001","Codex CLI · effort high · short fixed tasks","\u0001","\u0001","\u0001"],["cli-model-latency-tokens","scheduler-repair-claude-vs-codex","Total time","Codex CLI · GPT-6.1 Sol · medium","scheduler-repair-claude-vs-codex/Total time","Repairing a scheduler: Claude Code vs Codex vs API (Total time)",61.16,"seconds","61.2 s",3,"\u0001","minmax","Codex CLI · effort medium · scheduler repair, 296 checks, 3 runs",[59.9,69.51],"\u0001","\u0001"],["cli-model-latency-tokens","scheduler-repair-claude-vs-codex","First useful output","Codex CLI · GPT-6.1 Sol · medium","scheduler-repair-claude-vs-codex/First useful output","Repairing a scheduler: Claude Code vs Codex vs API (First useful output)",15.56,"seconds","15.6 s",3,"\u0001","minmax","Codex CLI · effort medium · scheduler repair, 296 checks, 3 runs",[13.65,23.04],"\u0001","\u0001"],["cli-model-latency-tokens","scheduler-repair-output-tokens","Output tokens","Codex CLI · GPT-6.1 Sol · medium","scheduler-repair-output-tokens/Output tokens","Output tokens to repair the scheduler (Output tokens)",1181,"tokens","1,181",3,"\u0001","\u0001","Codex CLI · effort medium · scheduler repair, 296 checks, 3 runs","\u0001","\u0001","\u0001"],["json-schema-vs-instructions","structured-output-pass-rate","Strict pass: the whole reply is the right JSON","GPT-6.1 Sol (low, instructions) · Codex CLI","structured-output-pass-rate/Strict pass: the whole reply is the right JSON","Does a JSON schema raise the pass rate? Instructions vs schema mode (Strict pass: the whole reply is the right JSON)",1,"rate","100% (12/12)",12,[0.7575,1],"ci95","Codex CLI · effort low · instructions","\u0001",true,"higher"],["json-schema-vs-instructions","structured-output-pass-rate","Strict pass: the whole reply is the right JSON","GPT-6.1 Sol (low, JSON schema) · Codex CLI","structured-output-pass-rate/Strict pass: the whole reply is the right JSON","Does a JSON schema raise the pass rate? Instructions vs schema mode (Strict pass: the whole reply is the right JSON)",1,"rate","100% (12/12)",12,[0.7575,1],"ci95","Codex CLI · effort low · JSON schema","\u0001",true,"higher"],["json-schema-vs-instructions","structured-output-pass-rate","Right answer in any format (strict pass or format miss)","GPT-6.1 Sol (low, instructions) · Codex CLI","structured-output-pass-rate/Right answer in any format (strict pass or format miss)","Does a JSON schema raise the pass rate? Instructions vs schema mode (Right answer in any format (strict pass or format miss))",1,"rate","100% (12/12)",12,[0.7575,1],"ci95","Codex CLI · effort low · instructions","\u0001",true,"higher"],["json-schema-vs-instructions","structured-output-pass-rate","Right answer in any format (strict pass or format miss)","GPT-6.1 Sol (low, JSON schema) · Codex CLI","structured-output-pass-rate/Right answer in any format (strict pass or format miss)","Does a JSON schema raise the pass rate? Instructions vs schema mode (Right answer in any format (strict pass or format miss))",1,"rate","100% (12/12)",12,[0.7575,1],"ci95","Codex CLI · effort low · JSON schema","\u0001",true,"higher"],["json-schema-vs-instructions","structured-output-outcomes","Strict pass","GPT-6.1 Sol (low, instructions) · Codex CLI","structured-output-outcomes/Strict pass","What each call produced: strict pass, format miss, wrong values or error (Strict pass)",12,"count","12",12,"\u0001","\u0001","Codex CLI · effort low · instructions","\u0001","\u0001","\u0001"],["json-schema-vs-instructions","structured-output-outcomes","Strict pass","GPT-6.1 Sol (low, JSON schema) · Codex CLI","structured-output-outcomes/Strict pass","What each call produced: strict pass, format miss, wrong values or error (Strict pass)",12,"count","12",12,"\u0001","\u0001","Codex CLI · effort low · JSON schema","\u0001","\u0001","\u0001"],["json-schema-vs-instructions","structured-output-outcomes","Format miss","GPT-6.1 Sol (low, instructions) · Codex CLI","structured-output-outcomes/Format miss","What each call produced: strict pass, format miss, wrong values or error (Format miss)",0,"count","0",12,"\u0001","\u0001","Codex CLI · effort low · instructions","\u0001","\u0001","\u0001"],["json-schema-vs-instructions","structured-output-outcomes","Format miss","GPT-6.1 Sol (low, JSON schema) · Codex CLI","structured-output-outcomes/Format miss","What each call produced: strict pass, format miss, wrong values or error (Format miss)",0,"count","0",12,"\u0001","\u0001","Codex CLI · effort low · JSON schema","\u0001","\u0001","\u0001"],["json-schema-vs-instructions","structured-output-outcomes","Wrong values","GPT-6.1 Sol (low, instructions) · Codex CLI","structured-output-outcomes/Wrong values","What each call produced: strict pass, format miss, wrong values or error (Wrong values)",0,"count","0",12,"\u0001","\u0001","Codex CLI · effort low · instructions","\u0001","\u0001","\u0001"],["json-schema-vs-instructions","structured-output-outcomes","Wrong values","GPT-6.1 Sol (low, JSON schema) · Codex CLI","structured-output-outcomes/Wrong values","What each call produced: strict pass, format miss, wrong values or error (Wrong values)",0,"count","0",12,"\u0001","\u0001","Codex CLI · effort low · JSON schema","\u0001","\u0001","\u0001"],["json-schema-vs-instructions","structured-output-outcomes","Error","GPT-6.1 Sol (low, instructions) · Codex CLI","structured-output-outcomes/Error","What each call produced: strict pass, format miss, wrong values or error (Error)",0,"count","0",12,"\u0001","\u0001","Codex CLI · effort low · instructions","\u0001","\u0001","\u0001"],["json-schema-vs-instructions","structured-output-outcomes","Error","GPT-6.1 Sol (low, JSON schema) · Codex CLI","structured-output-outcomes/Error","What each call produced: strict pass, format miss, wrong values or error (Error)",0,"count","0",12,"\u0001","\u0001","Codex CLI · effort low · JSON schema","\u0001","\u0001","\u0001"],["json-schema-vs-instructions","structured-output-time","Median time per call (the three prompts pooled)","GPT-6.1 Sol (low, instructions) · Codex CLI","structured-output-time","Time per call, instructions vs schema mode",6.21,"seconds","6.21 s",12,"\u0001","minmax","Codex CLI · effort low · instructions",[4.2,12.27],"\u0001","\u0001"],["json-schema-vs-instructions","structured-output-time","Median time per call (the three prompts pooled)","GPT-6.1 Sol (low, JSON schema) · Codex CLI","structured-output-time","Time per call, instructions vs schema mode",5.96,"seconds","5.96 s",12,"\u0001","minmax","Codex CLI · effort low · JSON schema",[4.62,20.97],"\u0001","\u0001"],["json-schema-vs-instructions","structured-output-tokens","Median output tokens per call","GPT-6.1 Sol (low, instructions) · Codex CLI","structured-output-tokens/Median output tokens per call","Output and reasoning tokens per call, instructions vs schema mode (Median output tokens per call)",117,"tokens","117",12,"\u0001","\u0001","Codex CLI · effort low · instructions","\u0001","\u0001","\u0001"],["json-schema-vs-instructions","structured-output-tokens","Median output tokens per call","GPT-6.1 Sol (low, JSON schema) · Codex CLI","structured-output-tokens/Median output tokens per call","Output and reasoning tokens per call, instructions vs schema mode (Median output tokens per call)",123,"tokens","123",12,"\u0001","\u0001","Codex CLI · effort low · JSON schema","\u0001","\u0001","\u0001"],["thinking-token-bill","thinking-bill-share","Median call","GPT-6.1 Sol (high) · Codex CLI","thinking-bill-share","Reasoning share of output tokens per call on hard tasks (calculation)",57.01,"percent","57%",16,"\u0001","minmax","Codex CLI · effort high",[29.19,90.8],true,"none"],["thinking-token-bill","thinking-bill-share","Median call","GPT-6.1 Sol (medium) · Codex CLI","thinking-bill-share","Reasoning share of output tokens per call on hard tasks (calculation)",46.33,"percent","46.3%",16,"\u0001","minmax","Codex CLI · effort medium",[11.42,86.85],true,"none"],["thinking-token-bill","thinking-bill-cost-per-call","Reasoning (output tokens)","GPT-6.1 Sol (medium) · Codex CLI","thinking-bill-cost-per-call/Reasoning (output tokens)","List-price cost per call: reasoning, remaining output and input (calculation) (Reasoning (output tokens))",0.002273,"usd","$0.0023",16,"\u0001","\u0001","Codex CLI · effort medium","\u0001",true,"none"],["thinking-token-bill","thinking-bill-cost-per-call","Reasoning (output tokens)","GPT-6.1 Sol (high) · Codex CLI","thinking-bill-cost-per-call/Reasoning (output tokens)","List-price cost per call: reasoning, remaining output and input (calculation) (Reasoning (output tokens))",0.004114,"usd","$0.0041",16,"\u0001","\u0001","Codex CLI · effort high","\u0001",true,"none"],["thinking-token-bill","thinking-bill-cost-per-call","Remaining output (visible-answer estimate)","GPT-6.1 Sol (medium) · Codex CLI","thinking-bill-cost-per-call/Remaining output (visible-answer estimate)","List-price cost per call: reasoning, remaining output and input (calculation) (Remaining output (visible-answer estimate))",0.003025,"usd","$0.0030",16,"\u0001","\u0001","Codex CLI · effort medium","\u0001",true,"none"],["thinking-token-bill","thinking-bill-cost-per-call","Remaining output (visible-answer estimate)","GPT-6.1 Sol (high) · Codex CLI","thinking-bill-cost-per-call/Remaining output (visible-answer estimate)","List-price cost per call: reasoning, remaining output and input (calculation) (Remaining output (visible-answer estimate))",0.002905,"usd","$0.0029",16,"\u0001","\u0001","Codex CLI · effort high","\u0001",true,"none"],["thinking-token-bill","thinking-bill-cost-per-call","Input (prompt, cache priced)","GPT-6.1 Sol (medium) · Codex CLI","thinking-bill-cost-per-call/Input (prompt, cache priced)","List-price cost per call: reasoning, remaining output and input (calculation) (Input (prompt, cache priced))",0.020339,"usd","$0.020",16,"\u0001","\u0001","Codex CLI · effort medium","\u0001",true,"none"],["thinking-token-bill","thinking-bill-cost-per-call","Input (prompt, cache priced)","GPT-6.1 Sol (high) · Codex CLI","thinking-bill-cost-per-call/Input (prompt, cache priced)","List-price cost per call: reasoning, remaining output and input (calculation) (Input (prompt, cache priced))",0.008117,"usd","$0.0081",16,"\u0001","\u0001","Codex CLI · effort high","\u0001",true,"none"],["thinking-token-bill","thinking-bill-by-effort","Reasoning cost per strict pass","GPT-6.1 Sol (low) · Codex CLI","thinking-bill-by-effort/Reasoning cost per strict pass","Reasoning cost per strict pass by effort, with the total (calculation) (Reasoning cost per strict pass)",0.001223,"usd","$0.0012",16,"\u0001","\u0001","Codex CLI · effort low","\u0001",true,"none"],["thinking-token-bill","thinking-bill-by-effort","Reasoning cost per strict pass","GPT-6.1 Sol (medium) · Codex CLI","thinking-bill-by-effort/Reasoning cost per strict pass","Reasoning cost per strict pass by effort, with the total (calculation) (Reasoning cost per strict pass)",0.002273,"usd","$0.0023",16,"\u0001","\u0001","Codex CLI · effort medium","\u0001",true,"none"],["thinking-token-bill","thinking-bill-by-effort","Reasoning cost per strict pass","GPT-6.1 Sol (high) · Codex CLI","thinking-bill-by-effort/Reasoning cost per strict pass","Reasoning cost per strict pass by effort, with the total (calculation) (Reasoning cost per strict pass)",0.004114,"usd","$0.0041",16,"\u0001","\u0001","Codex CLI · effort high","\u0001",true,"none"],["thinking-token-bill","thinking-bill-by-effort","Total cost per strict pass","GPT-6.1 Sol (low) · Codex CLI","thinking-bill-by-effort/Total cost per strict pass","Reasoning cost per strict pass by effort, with the total (calculation) (Total cost per strict pass)",0.012837,"usd","$0.013",16,"\u0001","\u0001","Codex CLI · effort low","\u0001",true,"none"],["thinking-token-bill","thinking-bill-by-effort","Total cost per strict pass","GPT-6.1 Sol (medium) · Codex CLI","thinking-bill-by-effort/Total cost per strict pass","Reasoning cost per strict pass by effort, with the total (calculation) (Total cost per strict pass)",0.025637,"usd","$0.026",16,"\u0001","\u0001","Codex CLI · effort medium","\u0001",true,"none"],["thinking-token-bill","thinking-bill-by-effort","Total cost per strict pass","GPT-6.1 Sol (high) · Codex CLI","thinking-bill-by-effort/Total cost per strict pass","Reasoning cost per strict pass by effort, with the total (calculation) (Total cost per strict pass)",0.015137,"usd","$0.015",16,"\u0001","\u0001","Codex CLI · effort high","\u0001",true,"none"],["thinking-token-bill","thinking-bill-short-vs-hard","Eight hard tasks","GPT-6.1 Sol (medium) · Codex CLI","thinking-bill-short-vs-hard/Eight hard tasks","Reasoning share on short tasks vs hard tasks (calculation) (Eight hard tasks)",46.33,"percent","46.3%",16,"\u0001","minmax","Codex CLI · effort medium",[11.42,86.85],true,"none"],["thinking-token-bill","thinking-bill-short-vs-hard","Eight hard tasks","GPT-6.1 Sol (high) · Codex CLI","thinking-bill-short-vs-hard/Eight hard tasks","Reasoning share on short tasks vs hard tasks (calculation) (Eight hard tasks)",57.01,"percent","57%",16,"\u0001","minmax","Codex CLI · effort high",[29.19,90.8],true,"none"],["thinking-token-bill","thinking-bill-short-vs-hard","Five short tasks","GPT-6.1 Sol (medium) · Codex CLI","thinking-bill-short-vs-hard/Five short tasks","Reasoning share on short tasks vs hard tasks (calculation) (Five short tasks)",41.05,"percent","41%",15,"\u0001","minmax","Codex CLI · effort medium",[0,71.43],true,"none"],["thinking-token-bill","thinking-bill-short-vs-hard","Five short tasks","GPT-6.1 Sol (high) · Codex CLI","thinking-bill-short-vs-hard/Five short tasks","Reasoning share on short tasks vs hard tasks (calculation) (Five short tasks)",58.06,"percent","58.1%",15,"\u0001","minmax","Codex CLI · effort high",[0,75.76],true,"none"],["llm-speed-anatomy","speed-anatomy-first-text","Time to first text","GPT-6.1 Sol (low) · Codex CLI","speed-anatomy-first-text","Time to first text: a 250-line answer, six models",3.52,"seconds","3.52 s",4,"\u0001","minmax","Codex CLI · effort low",[2.75,4.42],"\u0001","\u0001"],["llm-speed-anatomy","speed-anatomy-output-speed","Visible tokens per second","GPT-6.1 Sol (low) · Codex CLI","speed-anatomy-output-speed","Output speed after the first text: visible tokens per second (calculation)",79.6,"tokens","80",4,"\u0001","minmax","Codex CLI · effort low",[71.6,80.5],true,"\u0001"],["llm-speed-anatomy","speed-anatomy-chars-per-second","Characters per second","GPT-6.1 Sol (low) · Codex CLI","speed-anatomy-chars-per-second","Output speed in characters per second after the first text (calculation)",323,"count","323",4,"\u0001","minmax","Codex CLI · effort low",[291,327],true,"\u0001"],["llm-speed-anatomy","speed-anatomy-prompt-size","GPT-6.1 Sol (low) · Codex CLI","1k","speed-anatomy-prompt-size/1k","Time to first text as the prompt grows: 1k",3.36,"seconds","3.36 s",3,"\u0001","minmax","Codex CLI · effort low",[3.36,4.75],true,"\u0001"],["llm-speed-anatomy","speed-anatomy-prompt-size","GPT-6.1 Sol (low) · Codex CLI","16k","speed-anatomy-prompt-size/16k","Time to first text as the prompt grows: 16k",4.02,"seconds","4.02 s",3,"\u0001","minmax","Codex CLI · effort low",[3.3,4.28],true,"\u0001"],["llm-speed-anatomy","speed-anatomy-prompt-size","GPT-6.1 Sol (low) · Codex CLI","64k","speed-anatomy-prompt-size/64k","Time to first text as the prompt grows: 64k",3.93,"seconds","3.93 s",3,"\u0001","minmax","Codex CLI · effort low",[3.42,4.38],true,"\u0001"],["llm-speed-anatomy","speed-anatomy-total-by-size","1k prompt","GPT-6.1 Sol (low) · Codex CLI","speed-anatomy-total-by-size/1k prompt","Total time per call by prompt size (1k prompt)",3.43,"seconds","3.43 s",3,"\u0001","minmax","Codex CLI · effort low",[3.43,4.92],"\u0001","\u0001"],["llm-speed-anatomy","speed-anatomy-total-by-size","16k prompt","GPT-6.1 Sol (low) · Codex CLI","speed-anatomy-total-by-size/16k prompt","Total time per call by prompt size (16k prompt)",4.14,"seconds","4.14 s",3,"\u0001","minmax","Codex CLI · effort low",[3.96,4.68],"\u0001","\u0001"],["llm-speed-anatomy","speed-anatomy-total-by-size","64k prompt","GPT-6.1 Sol (low) · Codex CLI","speed-anatomy-total-by-size/64k prompt","Total time per call by prompt size (64k prompt)",3.96,"seconds","3.96 s",3,"\u0001","minmax","Codex CLI · effort low",[3.47,4.44],"\u0001","\u0001"],["llm-speed-anatomy","speed-anatomy-lookup-correct","Exact answer","GPT-6.1 Sol (low) · Codex CLI","speed-anatomy-lookup-correct","Exact lookup answers at the 1k, 16k and 64k prompt-size targets",1,"rate","100% (9/9)",9,[0.7009,1],"ci95","Codex CLI · effort low","\u0001","\u0001","higher"],["harder-tasks-head-to-head","harder-h2h-pass-rate","Strict pass","GPT-6.1 Sol (medium) · Codex CLI","harder-h2h-pass-rate/Strict pass","Pass rate on 4 harder tasks (Strict pass)",0.6875,"rate","69% (11/16)",16,[0.444,0.8584],"ci95","Codex CLI · effort medium","\u0001","\u0001","higher"],["harder-tasks-head-to-head","harder-h2h-pass-rate","Lenient (format misses counted)","GPT-6.1 Sol (medium) · Codex CLI","harder-h2h-pass-rate/Lenient (format misses counted)","Pass rate on 4 harder tasks (Lenient (format misses counted))",0.6875,"rate","69% (11/16)",16,[0.444,0.8584],"ci95","Codex CLI · effort medium","\u0001","\u0001","higher"],["harder-tasks-head-to-head","harder-h2h-tool-attempts","Tool attempt","GPT-6.1 Sol (medium) · Codex CLI","harder-h2h-tool-attempts","Calls that tried a tool although tools were off",0,"rate","0% (0/16)",16,[0,0.1936],"ci95","Codex CLI · effort medium","\u0001","\u0001","none"],["harder-tasks-head-to-head","harder-h2h-pass-by-task","GPT-6.1 Sol (medium) · Codex CLI","10x10 nonogram","harder-h2h-pass-by-task/10x10 nonogram","Strict pass rate by task: 10x10 nonogram",0.75,"rate","75% (3/4)",4,[0.3006,0.9544],"ci95","Codex CLI · effort medium","\u0001","\u0001","higher"],["harder-tasks-head-to-head","harder-h2h-pass-by-task","GPT-6.1 Sol (medium) · Codex CLI","Sudoku, 22 givens","harder-h2h-pass-by-task/Sudoku, 22 givens","Strict pass rate by task: Sudoku, 22 givens",0.25,"rate","25% (1/4)",4,[0.0456,0.6994],"ci95","Codex CLI · effort medium","\u0001","\u0001","higher"],["harder-tasks-head-to-head","harder-h2h-pass-by-task","GPT-6.1 Sol (medium) · Codex CLI","6x6 Skyscrapers","harder-h2h-pass-by-task/6x6 Skyscrapers","Strict pass rate by task: 6x6 Skyscrapers",1,"rate","100% (4/4)",4,[0.5101,1],"ci95","Codex CLI · effort medium","\u0001","\u0001","higher"],["harder-tasks-head-to-head","harder-h2h-pass-by-task","GPT-6.1 Sol (medium) · Codex CLI","Seeded shuffle output","harder-h2h-pass-by-task/Seeded shuffle output","Strict pass rate by task: Seeded shuffle output",0.75,"rate","75% (3/4)",4,[0.3006,0.9544],"ci95","Codex CLI · effort medium","\u0001","\u0001","higher"],["harder-tasks-head-to-head","harder-h2h-total-latency","Total time per call","GPT-6.1 Sol (medium) · Codex CLI","harder-h2h-total-latency","Total time per call on harder tasks",120.24,"seconds","120.2 s",13,"\u0001","minmax","Codex CLI · effort medium",[46.24,273.46],"\u0001","\u0001"],["harder-tasks-head-to-head","harder-h2h-output-tokens","Output tokens","GPT-6.1 Sol (medium) · Codex CLI","harder-h2h-output-tokens/Output tokens","Output tokens per call on harder tasks (Output tokens)",4994,"tokens","4,994",13,"\u0001","minmax","Codex CLI · effort medium",[2099,13413],"\u0001","none"],["harder-tasks-head-to-head","harder-h2h-cost-per-pass","Cost per strict pass","GPT-6.1 Sol (medium) · Codex CLI","harder-h2h-cost-per-pass","List-price cost per strict pass on harder tasks (calculation)",0.08293,"usd","$0.083",16,"\u0001","\u0001","Codex CLI · effort medium","\u0001",true,"\u0001"]]}],["claude-code-cli","Claude Code","Anthropic","cli","Anthropic’s coding CLI. Each measurement pairs it with one Claude model; the context names the model.",["Claude Code","Claude Code CLI"],{"$k":["studySlug","chartId","series","point","metric","label","value","unit","display","n","ci","spanKind","context","range","calculation","statId","polarity"],"$r":[["model-head-to-head","h2h-pass-rate","Pass rate","Claude Fable 5.1 · Claude Code","h2h-pass-rate","Pass rate on five validated tasks",1,"rate","100% (15/15)",15,[0.7961,1],"ci95","Claude Fable 5.1 · five short validated tasks","\u0001","\u0001","\u0001","\u0001"],["model-head-to-head","h2h-pass-rate","Pass rate","Claude Sonnet 5.5 · Claude Code","h2h-pass-rate","Pass rate on five validated tasks",0.8,"rate","80% (12/15)",15,[0.5481,0.9295],"ci95","Claude Sonnet 5.5 · five short validated tasks","\u0001","\u0001","\u0001","\u0001"],["model-head-to-head","h2h-pass-rate","Pass rate","Claude Opus 5.5 (high) · Claude Code","h2h-pass-rate","Pass rate on five validated tasks",1,"rate","100% (15/15)",15,[0.7961,1],"ci95","Claude Opus 5.5 · effort high · five short validated tasks","\u0001","\u0001","\u0001","\u0001"],["model-head-to-head","h2h-pass-rate","Pass rate","Claude Opus 5.5 · Claude Code","h2h-pass-rate","Pass rate on five validated tasks",1,"rate","100% (15/15)",15,[0.7961,1],"ci95","Claude Opus 5.5 · five short validated tasks","\u0001","\u0001","\u0001","\u0001"],["model-head-to-head","h2h-pass-rate","Pass rate","Claude Opus 5.5 (low) · Claude Code","h2h-pass-rate","Pass rate on five validated tasks",1,"rate","100% (15/15)",15,[0.7961,1],"ci95","Claude Opus 5.5 · effort low · five short validated tasks","\u0001","\u0001","\u0001","\u0001"],["model-head-to-head","h2h-pass-rate","Pass rate","Claude Haiku 4.5 · Claude Code","h2h-pass-rate","Pass rate on five validated tasks",1,"rate","100% (15/15)",15,[0.7961,1],"ci95","Claude Haiku 4.5 · five short validated tasks","\u0001","\u0001","\u0001","\u0001"],["model-head-to-head","h2h-total-latency","Total time per call","Claude Fable 5.1 · Claude Code","h2h-total-latency","Total time per call",1.94,"seconds","1.94 s",15,"\u0001","minmax","Claude Fable 5.1 · five short validated tasks",[1.41,9.83],"\u0001","\u0001","\u0001"],["model-head-to-head","h2h-total-latency","Total time per call","Claude Sonnet 5.5 · Claude Code","h2h-total-latency","Total time per call",2.31,"seconds","2.31 s",15,"\u0001","minmax","Claude Sonnet 5.5 · five short validated tasks",[2.17,7.73],"\u0001","\u0001","\u0001"],["model-head-to-head","h2h-total-latency","Total time per call","Claude Opus 5.5 (high) · Claude Code","h2h-total-latency","Total time per call",2.71,"seconds","2.71 s",15,"\u0001","minmax","Claude Opus 5.5 · effort high · five short validated tasks",[2.45,11.78],"\u0001","\u0001","\u0001"],["model-head-to-head","h2h-total-latency","Total time per call","Claude Opus 5.5 · Claude Code","h2h-total-latency","Total time per call",2.75,"seconds","2.75 s",15,"\u0001","minmax","Claude Opus 5.5 · five short validated tasks",[2.47,8.91],"\u0001","\u0001","\u0001"],["model-head-to-head","h2h-total-latency","Total time per call","Claude Opus 5.5 (low) · Claude Code","h2h-total-latency","Total time per call",2.83,"seconds","2.83 s",15,"\u0001","minmax","Claude Opus 5.5 · effort low · five short validated tasks",[2.35,6.62],"\u0001","\u0001","\u0001"],["model-head-to-head","h2h-total-latency","Total time per call","Claude Haiku 4.5 · Claude Code","h2h-total-latency","Total time per call",4.43,"seconds","4.43 s",15,"\u0001","minmax","Claude Haiku 4.5 · five short validated tasks",[3.16,23.57],"\u0001","\u0001","\u0001"],["model-head-to-head","h2h-first-useful-latency","Time to first useful output","Claude Fable 5.1 · Claude Code","h2h-first-useful-latency","Time to first useful output",1.2,"seconds","1.20 s",15,"\u0001","minmax","Claude Fable 5.1 · five short validated tasks",[0.95,7.9],"\u0001","\u0001","\u0001"],["model-head-to-head","h2h-first-useful-latency","Time to first useful output","Claude Sonnet 5.5 · Claude Code","h2h-first-useful-latency","Time to first useful output",1.56,"seconds","1.56 s",15,"\u0001","minmax","Claude Sonnet 5.5 · five short validated tasks",[0.99,6.39],"\u0001","\u0001","\u0001"],["model-head-to-head","h2h-first-useful-latency","Time to first useful output","Claude Opus 5.5 (high) · Claude Code","h2h-first-useful-latency","Time to first useful output",2.04,"seconds","2.04 s",15,"\u0001","minmax","Claude Opus 5.5 · effort high · five short validated tasks",[1.4,9.94],"\u0001","\u0001","\u0001"],["model-head-to-head","h2h-first-useful-latency","Time to first useful output","Claude Opus 5.5 · Claude Code","h2h-first-useful-latency","Time to first useful output",1.92,"seconds","1.92 s",15,"\u0001","minmax","Claude Opus 5.5 · five short validated tasks",[1.56,7.23],"\u0001","\u0001","\u0001"],["model-head-to-head","h2h-first-useful-latency","Time to first useful output","Claude Opus 5.5 (low) · Claude Code","h2h-first-useful-latency","Time to first useful output",2.39,"seconds","2.39 s",15,"\u0001","minmax","Claude Opus 5.5 · effort low · five short validated tasks",[1.45,4.9],"\u0001","\u0001","\u0001"],["model-head-to-head","h2h-first-useful-latency","Time to first useful output","Claude Haiku 4.5 · Claude Code","h2h-first-useful-latency","Time to first useful output",3.63,"seconds","3.63 s",15,"\u0001","minmax","Claude Haiku 4.5 · five short validated tasks",[2.78,22.27],"\u0001","\u0001","\u0001"],["model-head-to-head","h2h-input-tokens","Cache read","Claude Fable 5.1 · Claude Code","h2h-input-tokens/Cache read","Input tokens per call: what the CLI sends (Cache read)",2760,"tokens","2,760",15,"\u0001","\u0001","Claude Fable 5.1 · five short validated tasks","\u0001","\u0001","\u0001","\u0001"],["model-head-to-head","h2h-input-tokens","Cache read","Claude Sonnet 5.5 · Claude Code","h2h-input-tokens/Cache read","Input tokens per call: what the CLI sends (Cache read)",1401,"tokens","1,401",15,"\u0001","\u0001","Claude Sonnet 5.5 · five short validated tasks","\u0001","\u0001","\u0001","\u0001"],["model-head-to-head","h2h-input-tokens","Cache read","Claude Opus 5.5 (high) · Claude Code","h2h-input-tokens/Cache read","Input tokens per call: what the CLI sends (Cache read)",1463,"tokens","1,463",15,"\u0001","\u0001","Claude Opus 5.5 · effort high · five short validated tasks","\u0001","\u0001","\u0001","\u0001"],["model-head-to-head","h2h-input-tokens","Cache read","Claude Opus 5.5 · Claude Code","h2h-input-tokens/Cache read","Input tokens per call: what the CLI sends (Cache read)",1401,"tokens","1,401",15,"\u0001","\u0001","Claude Opus 5.5 · five short validated tasks","\u0001","\u0001","\u0001","\u0001"],["model-head-to-head","h2h-input-tokens","Cache read","Claude Opus 5.5 (low) · Claude Code","h2h-input-tokens/Cache read","Input tokens per call: what the CLI sends (Cache read)",1463,"tokens","1,463",15,"\u0001","\u0001","Claude Opus 5.5 · effort low · five short validated tasks","\u0001","\u0001","\u0001","\u0001"],["model-head-to-head","h2h-input-tokens","Cache read","Claude Haiku 4.5 · Claude Code","h2h-input-tokens/Cache read","Input tokens per call: what the CLI sends (Cache read)",0,"tokens","0",15,"\u0001","\u0001","Claude Haiku 4.5 · five short validated tasks","\u0001","\u0001","\u0001","\u0001"],["model-head-to-head","h2h-input-tokens","Other input","Claude Fable 5.1 · Claude Code","h2h-input-tokens/Other input","Input tokens per call: what the CLI sends (Other input)",473,"tokens","473",15,"\u0001","\u0001","Claude Fable 5.1 · five short validated tasks","\u0001","\u0001","\u0001","\u0001"],["model-head-to-head","h2h-input-tokens","Other input","Claude Sonnet 5.5 · Claude Code","h2h-input-tokens/Other input","Input tokens per call: what the CLI sends (Other input)",685,"tokens","685",15,"\u0001","\u0001","Claude Sonnet 5.5 · five short validated tasks","\u0001","\u0001","\u0001","\u0001"],["model-head-to-head","h2h-input-tokens","Other input","Claude Opus 5.5 (high) · Claude Code","h2h-input-tokens/Other input","Input tokens per call: what the CLI sends (Other input)",619,"tokens","619",15,"\u0001","\u0001","Claude Opus 5.5 · effort high · five short validated tasks","\u0001","\u0001","\u0001","\u0001"],["model-head-to-head","h2h-input-tokens","Other input","Claude Opus 5.5 · Claude Code","h2h-input-tokens/Other input","Input tokens per call: what the CLI sends (Other input)",680,"tokens","680",15,"\u0001","\u0001","Claude Opus 5.5 · five short validated tasks","\u0001","\u0001","\u0001","\u0001"],["model-head-to-head","h2h-input-tokens","Other input","Claude Opus 5.5 (low) · Claude Code","h2h-input-tokens/Other input","Input tokens per call: what the CLI sends (Other input)",618,"tokens","618",15,"\u0001","\u0001","Claude Opus 5.5 · effort low · five short validated tasks","\u0001","\u0001","\u0001","\u0001"],["model-head-to-head","h2h-input-tokens","Other input","Claude Haiku 4.5 · Claude Code","h2h-input-tokens/Other input","Input tokens per call: what the CLI sends (Other input)",3790,"tokens","3,790",15,"\u0001","\u0001","Claude Haiku 4.5 · five short validated tasks","\u0001","\u0001","\u0001","\u0001"],["model-head-to-head","h2h-output-tokens","Output tokens","Claude Fable 5.1 · Claude Code","h2h-output-tokens/Output tokens","Output tokens per call (Output tokens)",64,"tokens","64",15,"\u0001","\u0001","Claude Fable 5.1 · five short validated tasks","\u0001","\u0001","\u0001","\u0001"],["model-head-to-head","h2h-output-tokens","Output tokens","Claude Sonnet 5.5 · Claude Code","h2h-output-tokens/Output tokens","Output tokens per call (Output tokens)",107,"tokens","107",15,"\u0001","\u0001","Claude Sonnet 5.5 · five short validated tasks","\u0001","\u0001","\u0001","\u0001"],["model-head-to-head","h2h-output-tokens","Output tokens","Claude Opus 5.5 (high) · Claude Code","h2h-output-tokens/Output tokens","Output tokens per call (Output tokens)",78,"tokens","78",15,"\u0001","\u0001","Claude Opus 5.5 · effort high · five short validated tasks","\u0001","\u0001","\u0001","\u0001"],["model-head-to-head","h2h-output-tokens","Output tokens","Claude Opus 5.5 · Claude Code","h2h-output-tokens/Output tokens","Output tokens per call (Output tokens)",64,"tokens","64",15,"\u0001","\u0001","Claude Opus 5.5 · five short validated tasks","\u0001","\u0001","\u0001","\u0001"],["model-head-to-head","h2h-output-tokens","Output tokens","Claude Opus 5.5 (low) · Claude Code","h2h-output-tokens/Output tokens","Output tokens per call (Output tokens)",64,"tokens","64",15,"\u0001","\u0001","Claude Opus 5.5 · effort low · five short validated tasks","\u0001","\u0001","\u0001","\u0001"],["model-head-to-head","h2h-output-tokens","Output tokens","Claude Haiku 4.5 · Claude Code","h2h-output-tokens/Output tokens","Output tokens per call (Output tokens)",367,"tokens","367",15,"\u0001","\u0001","Claude Haiku 4.5 · five short validated tasks","\u0001","\u0001","\u0001","\u0001"],["model-head-to-head","h2h-list-price-per-call","Cost per call","Claude Fable 5.1 · Claude Code","h2h-list-price-per-call","List-price cost per call (calculation)",0.00987,"usd","$0.0099",15,"\u0001","minmax","Claude Fable 5.1 · five short validated tasks",[0.0049,0.05843],true,"\u0001","\u0001"],["model-head-to-head","h2h-list-price-per-call","Cost per call","Claude Sonnet 5.5 · Claude Code","h2h-list-price-per-call","List-price cost per call (calculation)",0.0036,"usd","$0.0036",15,"\u0001","minmax","Claude Sonnet 5.5 · five short validated tasks",[0.00342,0.01021],true,"\u0001","\u0001"],["model-head-to-head","h2h-list-price-per-call","Cost per call","Claude Opus 5.5 (high) · Claude Code","h2h-list-price-per-call","List-price cost per call (calculation)",0.00694,"usd","$0.0069",15,"\u0001","minmax","Claude Opus 5.5 · effort high · five short validated tasks",[0.00592,0.02708],true,"\u0001","\u0001"],["model-head-to-head","h2h-list-price-per-call","Cost per call","Claude Opus 5.5 · Claude Code","h2h-list-price-per-call","List-price cost per call (calculation)",0.00688,"usd","$0.0069",15,"\u0001","minmax","Claude Opus 5.5 · five short validated tasks",[0.00592,0.02226],true,"\u0001","\u0001"],["model-head-to-head","h2h-list-price-per-call","Cost per call","Claude Opus 5.5 (low) · Claude Code","h2h-list-price-per-call","List-price cost per call (calculation)",0.00688,"usd","$0.0069",15,"\u0001","minmax","Claude Opus 5.5 · effort low · five short validated tasks",[0.00582,0.01793],true,"\u0001","\u0001"],["model-head-to-head","h2h-list-price-per-call","Cost per call","Claude Haiku 4.5 · Claude Code","h2h-list-price-per-call","List-price cost per call (calculation)",0.00566,"usd","$0.0057",15,"\u0001","minmax","Claude Haiku 4.5 · five short validated tasks",[0.00513,0.01804],true,"\u0001","\u0001"],["model-head-to-head","h2h-cost-per-pass","Cost per pass","Claude Sonnet 5.5 · Claude Code","h2h-cost-per-pass","List-price cost per passing answer (calculation)",0.00624,"usd","$0.0062",15,"\u0001","\u0001","Claude Sonnet 5.5 · five short validated tasks","\u0001",true,"\u0001","\u0001"],["model-head-to-head","h2h-cost-per-pass","Cost per pass","Claude Opus 5.5 (low) · Claude Code","h2h-cost-per-pass","List-price cost per passing answer (calculation)",0.00829,"usd","$0.0083",15,"\u0001","\u0001","Claude Opus 5.5 · effort low · five short validated tasks","\u0001",true,"\u0001","\u0001"],["model-head-to-head","h2h-cost-per-pass","Cost per pass","Claude Haiku 4.5 · Claude Code","h2h-cost-per-pass","List-price cost per passing answer (calculation)",0.00836,"usd","$0.0084",15,"\u0001","\u0001","Claude Haiku 4.5 · five short validated tasks","\u0001",true,"\u0001","\u0001"],["model-head-to-head","h2h-cost-per-pass","Cost per pass","Claude Opus 5.5 · Claude Code","h2h-cost-per-pass","List-price cost per passing answer (calculation)",0.01009,"usd","$0.010",15,"\u0001","\u0001","Claude Opus 5.5 · five short validated tasks","\u0001",true,"\u0001","\u0001"],["model-head-to-head","h2h-cost-per-pass","Cost per pass","Claude Opus 5.5 (high) · Claude Code","h2h-cost-per-pass","List-price cost per passing answer (calculation)",0.01049,"usd","$0.010",15,"\u0001","\u0001","Claude Opus 5.5 · effort high · five short validated tasks","\u0001",true,"\u0001","\u0001"],["model-head-to-head","h2h-cost-per-pass","Cost per pass","Claude Fable 5.1 · Claude Code","h2h-cost-per-pass","List-price cost per passing answer (calculation)",0.02054,"usd","$0.021",15,"\u0001","\u0001","Claude Fable 5.1 · five short validated tasks","\u0001",true,"\u0001","\u0001"],["hard-model-head-to-head","hard-h2h-pass-rate","Strict pass","Claude Sonnet 5.5 · Claude Code","hard-h2h-pass-rate/Strict pass","Pass rate on eight hard tasks (Strict pass)",1,"rate","100% (24/24)",24,[0.862,1],"ci95","Claude Sonnet 5.5 · eight hard validated tasks","\u0001","\u0001","\u0001","\u0001"],["hard-model-head-to-head","hard-h2h-pass-rate","Strict pass","Claude Opus 5.5 · Claude Code","hard-h2h-pass-rate/Strict pass","Pass rate on eight hard tasks (Strict pass)",1,"rate","100% (24/24)",24,[0.862,1],"ci95","Claude Opus 5.5 · eight hard validated tasks","\u0001","\u0001","\u0001","\u0001"],["hard-model-head-to-head","hard-h2h-pass-rate","Strict pass","Claude Opus 5.5 (high) · Claude Code","hard-h2h-pass-rate/Strict pass","Pass rate on eight hard tasks (Strict pass)",1,"rate","100% (24/24)",24,[0.862,1],"ci95","Claude Opus 5.5 · effort high · eight hard validated tasks","\u0001","\u0001","\u0001","\u0001"],["hard-model-head-to-head","hard-h2h-pass-rate","Strict pass","Claude Fable 5.1 · Claude Code","hard-h2h-pass-rate/Strict pass","Pass rate on eight hard tasks (Strict pass)",1,"rate","100% (24/24)",24,[0.862,1],"ci95","Claude Fable 5.1 · eight hard validated tasks","\u0001","\u0001","\u0001","\u0001"],["hard-model-head-to-head","hard-h2h-pass-rate","Strict pass","Claude Haiku 4.5 · Claude Code","hard-h2h-pass-rate/Strict pass","Pass rate on eight hard tasks (Strict pass)",0.4583,"rate","46% (11/24)",24,[0.2789,0.6493],"ci95","Claude Haiku 4.5 · eight hard validated tasks","\u0001","\u0001","\u0001","\u0001"],["hard-model-head-to-head","hard-h2h-pass-rate","Lenient (format misses counted)","Claude Sonnet 5.5 · Claude Code","hard-h2h-pass-rate/Lenient (format misses counted)","Pass rate on eight hard tasks (Lenient (format misses counted))",1,"rate","100% (24/24)",24,[0.862,1],"ci95","Claude Sonnet 5.5 · eight hard validated tasks","\u0001","\u0001","\u0001","\u0001"],["hard-model-head-to-head","hard-h2h-pass-rate","Lenient (format misses counted)","Claude Opus 5.5 · Claude Code","hard-h2h-pass-rate/Lenient (format misses counted)","Pass rate on eight hard tasks (Lenient (format misses counted))",1,"rate","100% (24/24)",24,[0.862,1],"ci95","Claude Opus 5.5 · eight hard validated tasks","\u0001","\u0001","\u0001","\u0001"],["hard-model-head-to-head","hard-h2h-pass-rate","Lenient (format misses counted)","Claude Opus 5.5 (high) · Claude Code","hard-h2h-pass-rate/Lenient (format misses counted)","Pass rate on eight hard tasks (Lenient (format misses counted))",1,"rate","100% (24/24)",24,[0.862,1],"ci95","Claude Opus 5.5 · effort high · eight hard validated tasks","\u0001","\u0001","\u0001","\u0001"],["hard-model-head-to-head","hard-h2h-pass-rate","Lenient (format misses counted)","Claude Fable 5.1 · Claude Code","hard-h2h-pass-rate/Lenient (format misses counted)","Pass rate on eight hard tasks (Lenient (format misses counted))",1,"rate","100% (24/24)",24,[0.862,1],"ci95","Claude Fable 5.1 · eight hard validated tasks","\u0001","\u0001","\u0001","\u0001"],["hard-model-head-to-head","hard-h2h-pass-rate","Lenient (format misses counted)","Claude Haiku 4.5 · Claude Code","hard-h2h-pass-rate/Lenient (format misses counted)","Pass rate on eight hard tasks (Lenient (format misses counted))",0.6667,"rate","67% (16/24)",24,[0.4671,0.8203],"ci95","Claude Haiku 4.5 · eight hard validated tasks","\u0001","\u0001","\u0001","\u0001"],["hard-model-head-to-head","hard-h2h-total-latency","Total time per call on hard tasks (separate batches)","Claude Sonnet 5.5 · Claude Code","hard-h2h-total-latency","Total time per call on hard tasks (separate batches)",7.75,"seconds","7.75 s",24,"\u0001","minmax","Claude Sonnet 5.5 · eight hard validated tasks",[2.26,34.79],"\u0001","\u0001","\u0001"],["hard-model-head-to-head","hard-h2h-total-latency","Total time per call on hard tasks (separate batches)","Claude Opus 5.5 · Claude Code","hard-h2h-total-latency","Total time per call on hard tasks (separate batches)",9.18,"seconds","9.18 s",24,"\u0001","minmax","Claude Opus 5.5 · eight hard validated tasks",[4.24,27.21],"\u0001","\u0001","\u0001"],["hard-model-head-to-head","hard-h2h-total-latency","Total time per call on hard tasks (separate batches)","Claude Opus 5.5 (high) · Claude Code","hard-h2h-total-latency","Total time per call on hard tasks (separate batches)",11.03,"seconds","11.0 s",24,"\u0001","minmax","Claude Opus 5.5 · effort high · eight hard validated tasks",[3.63,63],"\u0001","\u0001","\u0001"],["hard-model-head-to-head","hard-h2h-total-latency","Total time per call on hard tasks (separate batches)","Claude Fable 5.1 · Claude Code","hard-h2h-total-latency","Total time per call on hard tasks (separate batches)",16.13,"seconds","16.1 s",24,"\u0001","minmax","Claude Fable 5.1 · eight hard validated tasks",[4.46,90],"\u0001","\u0001","\u0001"],["hard-model-head-to-head","hard-h2h-total-latency","Total time per call on hard tasks (separate batches)","Claude Haiku 4.5 · Claude Code","hard-h2h-total-latency","Total time per call on hard tasks (separate batches)",39.01,"seconds","39.0 s",24,"\u0001","minmax","Claude Haiku 4.5 · eight hard validated tasks",[15.27,75.13],"\u0001","\u0001","\u0001"],["hard-model-head-to-head","hard-h2h-first-useful-latency","Time to first useful output on hard tasks","Claude Sonnet 5.5 · Claude Code","hard-h2h-first-useful-latency","Time to first useful output on hard tasks",5.95,"seconds","5.95 s",24,"\u0001","minmax","Claude Sonnet 5.5 · eight hard validated tasks",[0.86,30.57],"\u0001","\u0001","\u0001"],["hard-model-head-to-head","hard-h2h-first-useful-latency","Time to first useful output on hard tasks","Claude Opus 5.5 · Claude Code","hard-h2h-first-useful-latency","Time to first useful output on hard tasks",6.78,"seconds","6.78 s",24,"\u0001","minmax","Claude Opus 5.5 · eight hard validated tasks",[2.39,21.77],"\u0001","\u0001","\u0001"],["hard-model-head-to-head","hard-h2h-first-useful-latency","Time to first useful output on hard tasks","Claude Opus 5.5 (high) · Claude Code","hard-h2h-first-useful-latency","Time to first useful output on hard tasks",7.13,"seconds","7.13 s",24,"\u0001","minmax","Claude Opus 5.5 · effort high · eight hard validated tasks",[2.15,56.23],"\u0001","\u0001","\u0001"],["hard-model-head-to-head","hard-h2h-first-useful-latency","Time to first useful output on hard tasks","Claude Fable 5.1 · Claude Code","hard-h2h-first-useful-latency","Time to first useful output on hard tasks",11.63,"seconds","11.6 s",24,"\u0001","minmax","Claude Fable 5.1 · eight hard validated tasks",[2,85.33],"\u0001","\u0001","\u0001"],["hard-model-head-to-head","hard-h2h-first-useful-latency","Time to first useful output on hard tasks","Claude Haiku 4.5 · Claude Code","hard-h2h-first-useful-latency","Time to first useful output on hard tasks",35.54,"seconds","35.5 s",24,"\u0001","minmax","Claude Haiku 4.5 · eight hard validated tasks",[12.88,70.31],"\u0001","\u0001","\u0001"],["hard-model-head-to-head","hard-h2h-output-tokens","Output tokens","Claude Sonnet 5.5 · Claude Code","hard-h2h-output-tokens/Output tokens","Output tokens per call on hard tasks (Output tokens)",1050,"tokens","1,050",24,"\u0001","\u0001","Claude Sonnet 5.5 · eight hard validated tasks","\u0001","\u0001","\u0001","\u0001"],["hard-model-head-to-head","hard-h2h-output-tokens","Output tokens","Claude Opus 5.5 · Claude Code","hard-h2h-output-tokens/Output tokens","Output tokens per call on hard tasks (Output tokens)",945,"tokens","945",24,"\u0001","\u0001","Claude Opus 5.5 · eight hard validated tasks","\u0001","\u0001","\u0001","\u0001"],["hard-model-head-to-head","hard-h2h-output-tokens","Output tokens","Claude Opus 5.5 (high) · Claude Code","hard-h2h-output-tokens/Output tokens","Output tokens per call on hard tasks (Output tokens)",1052,"tokens","1,052",24,"\u0001","\u0001","Claude Opus 5.5 · effort high · eight hard validated tasks","\u0001","\u0001","\u0001","\u0001"],["hard-model-head-to-head","hard-h2h-output-tokens","Output tokens","Claude Fable 5.1 · Claude Code","hard-h2h-output-tokens/Output tokens","Output tokens per call on hard tasks (Output tokens)",1366,"tokens","1,366",24,"\u0001","\u0001","Claude Fable 5.1 · eight hard validated tasks","\u0001","\u0001","\u0001","\u0001"],["hard-model-head-to-head","hard-h2h-output-tokens","Output tokens","Claude Haiku 4.5 · Claude Code","hard-h2h-output-tokens/Output tokens","Output tokens per call on hard tasks (Output tokens)",5064,"tokens","5,064",24,"\u0001","\u0001","Claude Haiku 4.5 · eight hard validated tasks","\u0001","\u0001","\u0001","\u0001"],["hard-model-head-to-head","hard-h2h-cost-per-pass","Cost per strict pass","Claude Sonnet 5.5 · Claude Code","hard-h2h-cost-per-pass","List-price cost per strict pass on hard tasks (calculation)",0.01435,"usd","$0.014",24,"\u0001","\u0001","Claude Sonnet 5.5 · eight hard validated tasks","\u0001",true,"\u0001","\u0001"],["hard-model-head-to-head","hard-h2h-cost-per-pass","Cost per strict pass","Claude Opus 5.5 · Claude Code","hard-h2h-cost-per-pass","List-price cost per strict pass on hard tasks (calculation)",0.02824,"usd","$0.028",24,"\u0001","\u0001","Claude Opus 5.5 · eight hard validated tasks","\u0001",true,"\u0001","\u0001"],["hard-model-head-to-head","hard-h2h-cost-per-pass","Cost per strict pass","Claude Opus 5.5 (high) · Claude Code","hard-h2h-cost-per-pass","List-price cost per strict pass on hard tasks (calculation)",0.03337,"usd","$0.033",24,"\u0001","\u0001","Claude Opus 5.5 · effort high · eight hard validated tasks","\u0001",true,"\u0001","\u0001"],["hard-model-head-to-head","hard-h2h-cost-per-pass","Cost per strict pass","Claude Haiku 4.5 · Claude Code","hard-h2h-cost-per-pass","List-price cost per strict pass on hard tasks (calculation)",0.0672,"usd","$0.067",24,"\u0001","\u0001","Claude Haiku 4.5 · eight hard validated tasks","\u0001",true,"\u0001","\u0001"],["hard-model-head-to-head","hard-h2h-cost-per-pass","Cost per strict pass","Claude Fable 5.1 · Claude Code","hard-h2h-cost-per-pass","List-price cost per strict pass on hard tasks (calculation)",0.09331,"usd","$0.093",24,"\u0001","\u0001","Claude Fable 5.1 · eight hard validated tasks","\u0001",true,"\u0001","\u0001"],["coding-agents-head-to-head","coding-agents-pass-rate","Passed every hidden check","Claude Sonnet 5.5 · Claude Code","coding-agents-pass-rate","Coding sessions that passed every hidden check",1,"rate","100% (12/12)",12,[0.7575,1],"ci95","Claude Sonnet 5.5 · six small repository tasks with hidden tests","\u0001","\u0001","\u0001","\u0001"],["coding-agents-head-to-head","coding-agents-pass-rate","Passed every hidden check","Claude Opus 5.5 · Claude Code","coding-agents-pass-rate","Coding sessions that passed every hidden check",1,"rate","100% (12/12)",12,[0.7575,1],"ci95","Claude Opus 5.5 · six small repository tasks with hidden tests","\u0001","\u0001","\u0001","\u0001"],["coding-agents-head-to-head","coding-agents-wall-time","Wall time per session","Claude Sonnet 5.5 · Claude Code","coding-agents-wall-time","Time per coding session",23.1,"seconds","23.1 s",12,"\u0001","minmax","Claude Sonnet 5.5 · six small repository tasks with hidden tests",[18.7,44.5],"\u0001","\u0001","\u0001"],["coding-agents-head-to-head","coding-agents-wall-time","Wall time per session","Claude Opus 5.5 · Claude Code","coding-agents-wall-time","Time per coding session",56.9,"seconds","56.9 s",12,"\u0001","minmax","Claude Opus 5.5 · six small repository tasks with hidden tests",[29.8,185.8],"\u0001","\u0001","\u0001"],["coding-agents-head-to-head","coding-agents-tool-calls","Tool calls per session","Claude Sonnet 5.5 · Claude Code","coding-agents-tool-calls","Tool calls per coding session",7.5,"calls","7.5",12,"\u0001","minmax","Claude Sonnet 5.5 · six small repository tasks with hidden tests",[3,14],"\u0001","\u0001","\u0001"],["coding-agents-head-to-head","coding-agents-tool-calls","Tool calls per session","Claude Opus 5.5 · Claude Code","coding-agents-tool-calls","Tool calls per coding session",7.5,"calls","7.5",12,"\u0001","minmax","Claude Opus 5.5 · six small repository tasks with hidden tests",[5,14],"\u0001","\u0001","\u0001"],["coding-agents-head-to-head","coding-agents-cost-per-pass","List-price cost per pass","Claude Sonnet 5.5 · Claude Code","coding-agents-cost-per-pass","List-price cost per passing coding session (calculation)",0.085,"usd","$0.085",12,"\u0001","\u0001","Claude Sonnet 5.5 · six small repository tasks with hidden tests","\u0001",true,"\u0001","\u0001"],["coding-agents-head-to-head","coding-agents-cost-per-pass","List-price cost per pass","Claude Opus 5.5 · Claude Code","coding-agents-cost-per-pass","List-price cost per passing coding session (calculation)",0.2229,"usd","$0.22",12,"\u0001","\u0001","Claude Opus 5.5 · six small repository tasks with hidden tests","\u0001",true,"\u0001","\u0001"],["effort-ladder","effort-ladder-pass-rate","Strict pass","Claude Sonnet 5.5 (low) · Claude Code","effort-ladder-pass-rate","Strict pass rate by effort on eight hard tasks",1,"rate","100% (16/16)",16,[0.8064,1],"ci95","Claude Sonnet 5.5 · effort low · eight hard validated tasks, effort ladder","\u0001","\u0001","\u0001","\u0001"],["effort-ladder","effort-ladder-pass-rate","Strict pass","Claude Sonnet 5.5 (medium) · Claude Code","effort-ladder-pass-rate","Strict pass rate by effort on eight hard tasks",1,"rate","100% (16/16)",16,[0.8064,1],"ci95","Claude Sonnet 5.5 · effort medium · eight hard validated tasks, effort ladder","\u0001","\u0001","\u0001","\u0001"],["effort-ladder","effort-ladder-pass-rate","Strict pass","Claude Sonnet 5.5 (high) · Claude Code","effort-ladder-pass-rate","Strict pass rate by effort on eight hard tasks",1,"rate","100% (16/16)",16,[0.8064,1],"ci95","Claude Sonnet 5.5 · effort high · eight hard validated tasks, effort ladder","\u0001","\u0001","\u0001","\u0001"],["effort-ladder","effort-ladder-pass-rate","Strict pass","Claude Sonnet 5.5 · Claude Code","effort-ladder-pass-rate","Strict pass rate by effort on eight hard tasks",1,"rate","100% (16/16)",16,[0.8064,1],"ci95","Claude Sonnet 5.5 · eight hard validated tasks, effort ladder","\u0001","\u0001","\u0001","\u0001"],["effort-ladder","effort-ladder-pass-rate","Strict pass","Claude Opus 5.5 (low) · Claude Code","effort-ladder-pass-rate","Strict pass rate by effort on eight hard tasks",1,"rate","100% (16/16)",16,[0.8064,1],"ci95","Claude Opus 5.5 · effort low · eight hard validated tasks, effort ladder","\u0001","\u0001","\u0001","\u0001"],["effort-ladder","effort-ladder-pass-rate","Strict pass","Claude Opus 5.5 (medium) · Claude Code","effort-ladder-pass-rate","Strict pass rate by effort on eight hard tasks",1,"rate","100% (16/16)",16,[0.8064,1],"ci95","Claude Opus 5.5 · effort medium · eight hard validated tasks, effort ladder","\u0001","\u0001","\u0001","\u0001"],["effort-ladder","effort-ladder-pass-rate","Strict pass","Claude Opus 5.5 (high) · Claude Code","effort-ladder-pass-rate","Strict pass rate by effort on eight hard tasks",1,"rate","100% (16/16)",16,[0.8064,1],"ci95","Claude Opus 5.5 · effort high · eight hard validated tasks, effort ladder","\u0001","\u0001","\u0001","\u0001"],["effort-ladder","effort-ladder-pass-rate","Strict pass","Claude Opus 5.5 · Claude Code","effort-ladder-pass-rate","Strict pass rate by effort on eight hard tasks",1,"rate","100% (16/16)",16,[0.8064,1],"ci95","Claude Opus 5.5 · eight hard validated tasks, effort ladder","\u0001","\u0001","\u0001","\u0001"],["effort-ladder","effort-ladder-total-latency","Total time per call by effort on hard tasks","Claude Sonnet 5.5 (low) · Claude Code","effort-ladder-total-latency","Total time per call by effort on hard tasks",5.82,"seconds","5.82 s",16,"\u0001","minmax","Claude Sonnet 5.5 · effort low · eight hard validated tasks, effort ladder",[2.78,19.96],"\u0001","\u0001","\u0001"],["effort-ladder","effort-ladder-total-latency","Total time per call by effort on hard tasks","Claude Sonnet 5.5 (medium) · Claude Code","effort-ladder-total-latency","Total time per call by effort on hard tasks",7.63,"seconds","7.63 s",16,"\u0001","minmax","Claude Sonnet 5.5 · effort medium · eight hard validated tasks, effort ladder",[2.71,24.01],"\u0001","\u0001","\u0001"],["effort-ladder","effort-ladder-total-latency","Total time per call by effort on hard tasks","Claude Sonnet 5.5 (high) · Claude Code","effort-ladder-total-latency","Total time per call by effort on hard tasks",8.81,"seconds","8.81 s",16,"\u0001","minmax","Claude Sonnet 5.5 · effort high · eight hard validated tasks, effort ladder",[2.93,35.81],"\u0001","\u0001","\u0001"],["effort-ladder","effort-ladder-total-latency","Total time per call by effort on hard tasks","Claude Sonnet 5.5 · Claude Code","effort-ladder-total-latency","Total time per call by effort on hard tasks",7.97,"seconds","7.97 s",16,"\u0001","minmax","Claude Sonnet 5.5 · eight hard validated tasks, effort ladder",[2.26,21.61],"\u0001","\u0001","\u0001"],["effort-ladder","effort-ladder-total-latency","Total time per call by effort on hard tasks","Claude Opus 5.5 (low) · Claude Code","effort-ladder-total-latency","Total time per call by effort on hard tasks",7.5,"seconds","7.50 s",16,"\u0001","minmax","Claude Opus 5.5 · effort low · eight hard validated tasks, effort ladder",[3.34,15.82],"\u0001","\u0001","\u0001"],["effort-ladder","effort-ladder-total-latency","Total time per call by effort on hard tasks","Claude Opus 5.5 (medium) · Claude Code","effort-ladder-total-latency","Total time per call by effort on hard tasks",9.72,"seconds","9.72 s",16,"\u0001","minmax","Claude Opus 5.5 · effort medium · eight hard validated tasks, effort ladder",[4.78,31.36],"\u0001","\u0001","\u0001"],["effort-ladder","effort-ladder-total-latency","Total time per call by effort on hard tasks","Claude Opus 5.5 (high) · Claude Code","effort-ladder-total-latency","Total time per call by effort on hard tasks",10.11,"seconds","10.1 s",16,"\u0001","minmax","Claude Opus 5.5 · effort high · eight hard validated tasks, effort ladder",[3.63,63],"\u0001","\u0001","\u0001"],["effort-ladder","effort-ladder-total-latency","Total time per call by effort on hard tasks","Claude Opus 5.5 · Claude Code","effort-ladder-total-latency","Total time per call by effort on hard tasks",9.18,"seconds","9.18 s",16,"\u0001","minmax","Claude Opus 5.5 · eight hard validated tasks, effort ladder",[4.24,27.21],"\u0001","\u0001","\u0001"],["effort-ladder","effort-ladder-output-tokens","Output tokens","Claude Sonnet 5.5 (low) · Claude Code","effort-ladder-output-tokens/Output tokens","Output tokens per call by effort on hard tasks (Output tokens)",667,"tokens","667",16,"\u0001","\u0001","Claude Sonnet 5.5 · effort low · eight hard validated tasks, effort ladder","\u0001","\u0001","\u0001","\u0001"],["effort-ladder","effort-ladder-output-tokens","Output tokens","Claude Sonnet 5.5 (medium) · Claude Code","effort-ladder-output-tokens/Output tokens","Output tokens per call by effort on hard tasks (Output tokens)",770,"tokens","770",16,"\u0001","\u0001","Claude Sonnet 5.5 · effort medium · eight hard validated tasks, effort ladder","\u0001","\u0001","\u0001","\u0001"],["effort-ladder","effort-ladder-output-tokens","Output tokens","Claude Sonnet 5.5 (high) · Claude Code","effort-ladder-output-tokens/Output tokens","Output tokens per call by effort on hard tasks (Output tokens)",1192,"tokens","1,192",16,"\u0001","\u0001","Claude Sonnet 5.5 · effort high · eight hard validated tasks, effort ladder","\u0001","\u0001","\u0001","\u0001"],["effort-ladder","effort-ladder-output-tokens","Output tokens","Claude Sonnet 5.5 · Claude Code","effort-ladder-output-tokens/Output tokens","Output tokens per call by effort on hard tasks (Output tokens)",1054,"tokens","1,054",16,"\u0001","\u0001","Claude Sonnet 5.5 · eight hard validated tasks, effort ladder","\u0001","\u0001","\u0001","\u0001"],["effort-ladder","effort-ladder-output-tokens","Output tokens","Claude Opus 5.5 (low) · Claude Code","effort-ladder-output-tokens/Output tokens","Output tokens per call by effort on hard tasks (Output tokens)",594,"tokens","594",16,"\u0001","\u0001","Claude Opus 5.5 · effort low · eight hard validated tasks, effort ladder","\u0001","\u0001","\u0001","\u0001"],["effort-ladder","effort-ladder-output-tokens","Output tokens","Claude Opus 5.5 (medium) · Claude Code","effort-ladder-output-tokens/Output tokens","Output tokens per call by effort on hard tasks (Output tokens)",853,"tokens","853",16,"\u0001","\u0001","Claude Opus 5.5 · effort medium · eight hard validated tasks, effort ladder","\u0001","\u0001","\u0001","\u0001"],["effort-ladder","effort-ladder-output-tokens","Output tokens","Claude Opus 5.5 (high) · Claude Code","effort-ladder-output-tokens/Output tokens","Output tokens per call by effort on hard tasks (Output tokens)",1052,"tokens","1,052",16,"\u0001","\u0001","Claude Opus 5.5 · effort high · eight hard validated tasks, effort ladder","\u0001","\u0001","\u0001","\u0001"],["effort-ladder","effort-ladder-output-tokens","Output tokens","Claude Opus 5.5 · Claude Code","effort-ladder-output-tokens/Output tokens","Output tokens per call by effort on hard tasks (Output tokens)",945,"tokens","945",16,"\u0001","\u0001","Claude Opus 5.5 · eight hard validated tasks, effort ladder","\u0001","\u0001","\u0001","\u0001"],["effort-ladder","effort-ladder-cost-per-pass","Cost per strict pass","Claude Sonnet 5.5 (low) · Claude Code","effort-ladder-cost-per-pass","List-price cost per strict pass by effort (calculation)",0.01219,"usd","$0.012",16,"\u0001","\u0001","Claude Sonnet 5.5 · effort low · eight hard validated tasks, effort ladder","\u0001",true,"\u0001","\u0001"],["effort-ladder","effort-ladder-cost-per-pass","Cost per strict pass","Claude Sonnet 5.5 (medium) · Claude Code","effort-ladder-cost-per-pass","List-price cost per strict pass by effort (calculation)",0.01352,"usd","$0.014",16,"\u0001","\u0001","Claude Sonnet 5.5 · effort medium · eight hard validated tasks, effort ladder","\u0001",true,"\u0001","\u0001"],["effort-ladder","effort-ladder-cost-per-pass","Cost per strict pass","Claude Sonnet 5.5 (high) · Claude Code","effort-ladder-cost-per-pass","List-price cost per strict pass by effort (calculation)",0.01671,"usd","$0.017",16,"\u0001","\u0001","Claude Sonnet 5.5 · effort high · eight hard validated tasks, effort ladder","\u0001",true,"\u0001","\u0001"],["effort-ladder","effort-ladder-cost-per-pass","Cost per strict pass","Claude Sonnet 5.5 · Claude Code","effort-ladder-cost-per-pass","List-price cost per strict pass by effort (calculation)",0.01398,"usd","$0.014",16,"\u0001","\u0001","Claude Sonnet 5.5 · eight hard validated tasks, effort ladder","\u0001",true,"\u0001","\u0001"],["effort-ladder","effort-ladder-cost-per-pass","Cost per strict pass","Claude Opus 5.5 (low) · Claude Code","effort-ladder-cost-per-pass","List-price cost per strict pass by effort (calculation)",0.02115,"usd","$0.021",16,"\u0001","\u0001","Claude Opus 5.5 · effort low · eight hard validated tasks, effort ladder","\u0001",true,"\u0001","\u0001"],["effort-ladder","effort-ladder-cost-per-pass","Cost per strict pass","Claude Opus 5.5 (medium) · Claude Code","effort-ladder-cost-per-pass","List-price cost per strict pass by effort (calculation)",0.02947,"usd","$0.029",16,"\u0001","\u0001","Claude Opus 5.5 · effort medium · eight hard validated tasks, effort ladder","\u0001",true,"\u0001","\u0001"],["effort-ladder","effort-ladder-cost-per-pass","Cost per strict pass","Claude Opus 5.5 (high) · Claude Code","effort-ladder-cost-per-pass","List-price cost per strict pass by effort (calculation)",0.03368,"usd","$0.034",16,"\u0001","\u0001","Claude Opus 5.5 · effort high · eight hard validated tasks, effort ladder","\u0001",true,"\u0001","\u0001"],["effort-ladder","effort-ladder-cost-per-pass","Cost per strict pass","Claude Opus 5.5 · Claude Code","effort-ladder-cost-per-pass","List-price cost per strict pass by effort (calculation)",0.02893,"usd","$0.029",16,"\u0001","\u0001","Claude Opus 5.5 · eight hard validated tasks, effort ladder","\u0001",true,"\u0001","\u0001"],["caching-consistency","caching-cost-with-without","With the cache, as recorded","Claude Sonnet 5.5 · Claude Code","caching-cost-with-without/With the cache, as recorded","List-price cost of 5-question sessions with and without the cache (calculation) (With the cache, as recorded)",0.135003,"usd","$0.14",15,"\u0001","\u0001","Claude Sonnet 5.5 · calculation: 5-turn cached sessions over a fixed ledger","\u0001",true,"\u0001","\u0001"],["caching-consistency","caching-cost-with-without","With the cache, as recorded","Claude Opus 5.5 · Claude Code","caching-cost-with-without/With the cache, as recorded","List-price cost of 5-question sessions with and without the cache (calculation) (With the cache, as recorded)",0.255057,"usd","$0.26",15,"\u0001","\u0001","Claude Opus 5.5 · calculation: 5-turn cached sessions over a fixed ledger","\u0001",true,"\u0001","\u0001"],["caching-consistency","caching-cost-with-without","Without a cache: every input token at the input price","Claude Sonnet 5.5 · Claude Code","caching-cost-with-without/Without a cache: every input token at the input price","List-price cost of 5-question sessions with and without the cache (calculation) (Without a cache: every input token at the input price)",0.269788,"usd","$0.27",15,"\u0001","\u0001","Claude Sonnet 5.5 · calculation: 5-turn cached sessions over a fixed ledger","\u0001",true,"\u0001","\u0001"],["caching-consistency","caching-cost-with-without","Without a cache: every input token at the input price","Claude Opus 5.5 · Claude Code","caching-cost-with-without/Without a cache: every input token at the input price","List-price cost of 5-question sessions with and without the cache (calculation) (Without a cache: every input token at the input price)",0.544228,"usd","$0.54",15,"\u0001","\u0001","Claude Opus 5.5 · calculation: 5-turn cached sessions over a fixed ledger","\u0001",true,"\u0001","\u0001"],["caching-consistency","caching-latency-first-vs-later","Turn 1 (writes the ledger to the cache)","Claude Sonnet 5.5 · Claude Code","caching-latency-first-vs-later/Turn 1 (writes the ledger to the cache)","Time per turn: first turn vs later turns in a cached session (Turn 1 (writes the ledger to the cache))",1.64,"seconds","1.64 s",3,"\u0001","minmax","Claude Sonnet 5.5 · 5-turn cached sessions over a fixed ledger",[1.58,1.79],"\u0001","\u0001","\u0001"],["caching-consistency","caching-latency-first-vs-later","Turn 1 (writes the ledger to the cache)","Claude Opus 5.5 · Claude Code","caching-latency-first-vs-later/Turn 1 (writes the ledger to the cache)","Time per turn: first turn vs later turns in a cached session (Turn 1 (writes the ledger to the cache))",1.9,"seconds","1.90 s",3,"\u0001","minmax","Claude Opus 5.5 · 5-turn cached sessions over a fixed ledger",[1.78,4.36],"\u0001","\u0001","\u0001"],["caching-consistency","caching-latency-first-vs-later","Turns 2-5 (read the ledger from the cache)","Claude Sonnet 5.5 · Claude Code","caching-latency-first-vs-later/Turns 2-5 (read the ledger from the cache)","Time per turn: first turn vs later turns in a cached session (Turns 2-5 (read the ledger from the cache))",1.61,"seconds","1.61 s",12,"\u0001","minmax","Claude Sonnet 5.5 · 5-turn cached sessions over a fixed ledger",[1.35,5.63],"\u0001","\u0001","\u0001"],["caching-consistency","caching-latency-first-vs-later","Turns 2-5 (read the ledger from the cache)","Claude Opus 5.5 · Claude Code","caching-latency-first-vs-later/Turns 2-5 (read the ledger from the cache)","Time per turn: first turn vs later turns in a cached session (Turns 2-5 (read the ledger from the cache))",2.4,"seconds","2.40 s",12,"\u0001","minmax","Claude Opus 5.5 · 5-turn cached sessions over a fixed ledger",[1.63,12.67],"\u0001","\u0001","\u0001"],["caching-consistency","consistency-pass-rate","Exact number","Claude Haiku 4.5 · Claude Code","consistency-pass-rate/Exact number","Same prompt, 10 times: strict pass rate (Exact number)",0,"rate","0% (0/10)",10,[0,0.2775],"ci95","Claude Haiku 4.5 · same prompt repeated 10 times","\u0001","\u0001","\u0001","\u0001"],["caching-consistency","consistency-pass-rate","Exact number","Claude Sonnet 5.5 · Claude Code","consistency-pass-rate/Exact number","Same prompt, 10 times: strict pass rate (Exact number)",1,"rate","100% (10/10)",10,[0.7225,1],"ci95","Claude Sonnet 5.5 · same prompt repeated 10 times","\u0001","\u0001","\u0001","\u0001"],["caching-consistency","consistency-pass-rate","JSON object","Claude Haiku 4.5 · Claude Code","consistency-pass-rate/JSON object","Same prompt, 10 times: strict pass rate (JSON object)",0.1,"rate","10% (1/10)",10,[0.0179,0.4042],"ci95","Claude Haiku 4.5 · same prompt repeated 10 times","\u0001","\u0001","\u0001","\u0001"],["caching-consistency","consistency-pass-rate","JSON object","Claude Sonnet 5.5 · Claude Code","consistency-pass-rate/JSON object","Same prompt, 10 times: strict pass rate (JSON object)",1,"rate","100% (10/10)",10,[0.7225,1],"ci95","Claude Sonnet 5.5 · same prompt repeated 10 times","\u0001","\u0001","\u0001","\u0001"],["caching-consistency","consistency-pass-rate","Code fix","Claude Haiku 4.5 · Claude Code","consistency-pass-rate/Code fix","Same prompt, 10 times: strict pass rate (Code fix)",1,"rate","100% (10/10)",10,[0.7225,1],"ci95","Claude Haiku 4.5 · same prompt repeated 10 times","\u0001","\u0001","\u0001","\u0001"],["caching-consistency","consistency-pass-rate","Code fix","Claude Sonnet 5.5 · Claude Code","consistency-pass-rate/Code fix","Same prompt, 10 times: strict pass rate (Code fix)",1,"rate","100% (10/10)",10,[0.7225,1],"ci95","Claude Sonnet 5.5 · same prompt repeated 10 times","\u0001","\u0001","\u0001","\u0001"],["caching-consistency","consistency-distinct-answers","Exact number","Claude Haiku 4.5 · Claude Code","consistency-distinct-answers/Exact number","Same prompt, 10 times: how many different answers (Exact number)",1,"count","1",10,"\u0001","\u0001","Claude Haiku 4.5 · same prompt repeated 10 times","\u0001","\u0001","\u0001","\u0001"],["caching-consistency","consistency-distinct-answers","Exact number","Claude Sonnet 5.5 · Claude Code","consistency-distinct-answers/Exact number","Same prompt, 10 times: how many different answers (Exact number)",1,"count","1",10,"\u0001","\u0001","Claude Sonnet 5.5 · same prompt repeated 10 times","\u0001","\u0001","\u0001","\u0001"],["caching-consistency","consistency-distinct-answers","JSON object","Claude Haiku 4.5 · Claude Code","consistency-distinct-answers/JSON object","Same prompt, 10 times: how many different answers (JSON object)",1,"count","1",10,"\u0001","\u0001","Claude Haiku 4.5 · same prompt repeated 10 times","\u0001","\u0001","\u0001","\u0001"],["caching-consistency","consistency-distinct-answers","JSON object","Claude Sonnet 5.5 · Claude Code","consistency-distinct-answers/JSON object","Same prompt, 10 times: how many different answers (JSON object)",1,"count","1",10,"\u0001","\u0001","Claude Sonnet 5.5 · same prompt repeated 10 times","\u0001","\u0001","\u0001","\u0001"],["caching-consistency","consistency-distinct-answers","Code fix","Claude Haiku 4.5 · Claude Code","consistency-distinct-answers/Code fix","Same prompt, 10 times: how many different answers (Code fix)",6,"count","6",10,"\u0001","\u0001","Claude Haiku 4.5 · same prompt repeated 10 times","\u0001","\u0001","\u0001","\u0001"],["caching-consistency","consistency-distinct-answers","Code fix","Claude Sonnet 5.5 · Claude Code","consistency-distinct-answers/Code fix","Same prompt, 10 times: how many different answers (Code fix)",3,"count","3",10,"\u0001","\u0001","Claude Sonnet 5.5 · same prompt repeated 10 times","\u0001","\u0001","\u0001","\u0001"],["caching-consistency","consistency-latency-spread","Exact number","Claude Haiku 4.5 · Claude Code","consistency-latency-spread/Exact number","Same prompt, 10 times: time per call (Exact number)",5.06,"seconds","5.06 s",10,"\u0001","minmax","Claude Haiku 4.5 · same prompt repeated 10 times",[4.42,6.2],"\u0001","\u0001","\u0001"],["caching-consistency","consistency-latency-spread","Exact number","Claude Sonnet 5.5 · Claude Code","consistency-latency-spread/Exact number","Same prompt, 10 times: time per call (Exact number)",6.89,"seconds","6.89 s",10,"\u0001","minmax","Claude Sonnet 5.5 · same prompt repeated 10 times",[5.81,7.81],"\u0001","\u0001","\u0001"],["caching-consistency","consistency-latency-spread","JSON object","Claude Haiku 4.5 · Claude Code","consistency-latency-spread/JSON object","Same prompt, 10 times: time per call (JSON object)",7.03,"seconds","7.03 s",10,"\u0001","minmax","Claude Haiku 4.5 · same prompt repeated 10 times",[5.28,12.27],"\u0001","\u0001","\u0001"],["caching-consistency","consistency-latency-spread","JSON object","Claude Sonnet 5.5 · Claude Code","consistency-latency-spread/JSON object","Same prompt, 10 times: time per call (JSON object)",2.89,"seconds","2.89 s",10,"\u0001","minmax","Claude Sonnet 5.5 · same prompt repeated 10 times",[2.68,5.3],"\u0001","\u0001","\u0001"],["caching-consistency","consistency-latency-spread","Code fix","Claude Haiku 4.5 · Claude Code","consistency-latency-spread/Code fix","Same prompt, 10 times: time per call (Code fix)",5.95,"seconds","5.95 s",10,"\u0001","minmax","Claude Haiku 4.5 · same prompt repeated 10 times",[4.89,7.33],"\u0001","\u0001","\u0001"],["caching-consistency","consistency-latency-spread","Code fix","Claude Sonnet 5.5 · Claude Code","consistency-latency-spread/Code fix","Same prompt, 10 times: time per call (Code fix)",2.67,"seconds","2.67 s",10,"\u0001","minmax","Claude Sonnet 5.5 · same prompt repeated 10 times",[2.32,4.34],"\u0001","\u0001","\u0001"],["agent-memory","\u0001","\u0001","\u0001","stat:memory-sessions","Claude Code sessions, every one graded (120 Sonnet 5.5, 80 Haiku 4.5)",200,"count","200",200,"\u0001","\u0001","","\u0001","\u0001","memory-sessions","\u0001"],["routing-overhead","cli-startup-tax","First output event","Claude Code · Claude Haiku 4.5","cli-startup-tax/First output event","CLI start-up tax on a one-word answer (First output event)",563,"ms","563 ms",5,"\u0001","minmax","Claude Haiku 4.5 · CLI start-up, one-word prompt, 5 runs",[519,726],"\u0001","\u0001","\u0001"],["routing-overhead","cli-startup-tax","First model output","Claude Code · Claude Haiku 4.5","cli-startup-tax/First model output","CLI start-up tax on a one-word answer (First model output)",1461,"ms","1,461 ms",5,"\u0001","minmax","Claude Haiku 4.5 · CLI start-up, one-word prompt, 5 runs",[1206,2308],"\u0001","\u0001","\u0001"],["routing-overhead","cli-startup-tax","Total wall time","Claude Code · Claude Haiku 4.5","cli-startup-tax/Total wall time","CLI start-up tax on a one-word answer (Total wall time)",2529,"ms","2,529 ms",5,"\u0001","minmax","Claude Haiku 4.5 · CLI start-up, one-word prompt, 5 runs",[2273,3382],"\u0001","\u0001","\u0001"],["routing-overhead","cli-startup-input-tokens","Input tokens per call","Claude Code · Claude Haiku 4.5","cli-startup-input-tokens","Input tokens a CLI sends for a one-word answer",6761,"tokens","6,761",5,"\u0001","\u0001","Claude Haiku 4.5 · CLI start-up, one-word prompt, 5 runs","\u0001","\u0001","\u0001","\u0001"],["routing-overhead","\u0001","\u0001","\u0001","stat:cli-startup-claude-harness-ms","Claude Code time outside the model on a one-word answer",1690,"ms","1,690 ms median (1,533 to 1,811)",5,"\u0001","\u0001","routing overhead per decision","\u0001","\u0001","cli-startup-claude-harness-ms","\u0001"],["cli-model-latency-tokens","scheduler-repair-claude-vs-codex","Total time","Claude Code CLI · Sonnet 5.5 · medium","scheduler-repair-claude-vs-codex/Total time","Repairing a scheduler: Claude Code vs Codex vs API (Total time)",15,"seconds","15.0 s",3,"\u0001","minmax","Sonnet 5.5 · effort medium · scheduler repair, 296 checks, 3 runs",[13.89,15.89],"\u0001","\u0001","\u0001"],["cli-model-latency-tokens","scheduler-repair-claude-vs-codex","First useful output","Claude Code CLI · Sonnet 5.5 · medium","scheduler-repair-claude-vs-codex/First useful output","Repairing a scheduler: Claude Code vs Codex vs API (First useful output)",7.55,"seconds","7.55 s",3,"\u0001","minmax","Sonnet 5.5 · effort medium · scheduler repair, 296 checks, 3 runs",[6.77,7.63],"\u0001","\u0001","\u0001"],["cli-model-latency-tokens","scheduler-repair-output-tokens","Output tokens","Claude Code CLI · Sonnet 5.5 · medium","scheduler-repair-output-tokens/Output tokens","Output tokens to repair the scheduler (Output tokens)",2227,"tokens","2,227",3,"\u0001","\u0001","Sonnet 5.5 · effort medium · scheduler repair, 296 checks, 3 runs","\u0001","\u0001","\u0001","\u0001"],["single-call-vs-agent-loop","agent-loop-pass-rate","Strict pass","Claude Haiku 4.5 (single call) · Claude Code","agent-loop-pass-rate","Strict pass rate: single call vs agent loop on eight hard tasks",0.4583,"rate","46% (11/24)",24,[0.2789,0.6493],"ci95","Claude Haiku 4.5 · single call","\u0001","\u0001","\u0001","higher"],["single-call-vs-agent-loop","agent-loop-pass-rate","Strict pass","Claude Haiku 4.5 (agent loop) · Claude Code","agent-loop-pass-rate","Strict pass rate: single call vs agent loop on eight hard tasks",0.5417,"rate","54% (13/24)",24,[0.3507,0.7211],"ci95","Claude Haiku 4.5 · agent loop","\u0001","\u0001","\u0001","higher"],["single-call-vs-agent-loop","agent-loop-pass-rate","Strict pass","Claude Sonnet 5.5 (single call) · Claude Code","agent-loop-pass-rate","Strict pass rate: single call vs agent loop on eight hard tasks",1,"rate","100% (24/24)",24,[0.862,1],"ci95","Claude Sonnet 5.5 · single call","\u0001","\u0001","\u0001","higher"],["single-call-vs-agent-loop","agent-loop-pass-rate","Strict pass","Claude Sonnet 5.5 (agent loop) · Claude Code","agent-loop-pass-rate","Strict pass rate: single call vs agent loop on eight hard tasks",1,"rate","100% (16/16)",16,[0.8064,1],"ci95","Claude Sonnet 5.5 · agent loop","\u0001","\u0001","\u0001","higher"],["single-call-vs-agent-loop","agent-loop-by-task","Claude Haiku 4.5 (single call) · Claude Code","Interval merge fix","agent-loop-by-task/Interval merge fix","Strict passes per task: single call vs agent loop: Interval merge fix",1,"rate","100% (3/3)",3,[0.4385,1],"ci95","Claude Haiku 4.5 · single call","\u0001","\u0001","\u0001","higher"],["single-call-vs-agent-loop","agent-loop-by-task","Claude Haiku 4.5 (single call) · Claude Code","DST day-length fix","agent-loop-by-task/DST day-length fix","Strict passes per task: single call vs agent loop: DST day-length fix",0.3333,"rate","33% (1/3)",3,[0.0615,0.7923],"ci95","Claude Haiku 4.5 · single call","\u0001","\u0001","\u0001","higher"],["single-call-vs-agent-loop","agent-loop-by-task","Claude Haiku 4.5 (single call) · Claude Code","CSV parser","agent-loop-by-task/CSV parser","Strict passes per task: single call vs agent loop: CSV parser",0.6667,"rate","67% (2/3)",3,[0.2077,0.9385],"ci95","Claude Haiku 4.5 · single call","\u0001","\u0001","\u0001","higher"],["single-call-vs-agent-loop","agent-loop-by-task","Claude Haiku 4.5 (single call) · Claude Code","Event-loop order","agent-loop-by-task/Event-loop order","Strict passes per task: single call vs agent loop: Event-loop order",0,"rate","0% (0/3)",3,[0,0.5615],"ci95","Claude Haiku 4.5 · single call","\u0001","\u0001","\u0001","higher"],["single-call-vs-agent-loop","agent-loop-by-task","Claude Haiku 4.5 (single call) · Claude Code","Room schedule","agent-loop-by-task/Room schedule","Strict passes per task: single call vs agent loop: Room schedule",0,"rate","0% (0/3)",3,[0,0.5615],"ci95","Claude Haiku 4.5 · single call","\u0001","\u0001","\u0001","higher"],["single-call-vs-agent-loop","agent-loop-by-task","Claude Haiku 4.5 (single call) · Claude Code","SemVer regex","agent-loop-by-task/SemVer regex","Strict passes per task: single call vs agent loop: SemVer regex",1,"rate","100% (3/3)",3,[0.4385,1],"ci95","Claude Haiku 4.5 · single call","\u0001","\u0001","\u0001","higher"],["single-call-vs-agent-loop","agent-loop-by-task","Claude Haiku 4.5 (single call) · Claude Code","Money refactor","agent-loop-by-task/Money refactor","Strict passes per task: single call vs agent loop: Money refactor",0.6667,"rate","67% (2/3)",3,[0.2077,0.9385],"ci95","Claude Haiku 4.5 · single call","\u0001","\u0001","\u0001","higher"],["single-call-vs-agent-loop","agent-loop-by-task","Claude Haiku 4.5 (single call) · Claude Code","SQL report","agent-loop-by-task/SQL report","Strict passes per task: single call vs agent loop: SQL report",0,"rate","0% (0/3)",3,[0,0.5615],"ci95","Claude Haiku 4.5 · single call","\u0001","\u0001","\u0001","higher"],["single-call-vs-agent-loop","agent-loop-by-task","Claude Haiku 4.5 (agent loop) · Claude Code","Interval merge fix","agent-loop-by-task/Interval merge fix","Strict passes per task: single call vs agent loop: Interval merge fix",0.6667,"rate","67% (2/3)",3,[0.2077,0.9385],"ci95","Claude Haiku 4.5 · agent loop","\u0001","\u0001","\u0001","higher"],["single-call-vs-agent-loop","agent-loop-by-task","Claude Haiku 4.5 (agent loop) · Claude Code","DST day-length fix","agent-loop-by-task/DST day-length fix","Strict passes per task: single call vs agent loop: DST day-length fix",0.6667,"rate","67% (2/3)",3,[0.2077,0.9385],"ci95","Claude Haiku 4.5 · agent loop","\u0001","\u0001","\u0001","higher"],["single-call-vs-agent-loop","agent-loop-by-task","Claude Haiku 4.5 (agent loop) · Claude Code","CSV parser","agent-loop-by-task/CSV parser","Strict passes per task: single call vs agent loop: CSV parser",0.6667,"rate","67% (2/3)",3,[0.2077,0.9385],"ci95","Claude Haiku 4.5 · agent loop","\u0001","\u0001","\u0001","higher"],["single-call-vs-agent-loop","agent-loop-by-task","Claude Haiku 4.5 (agent loop) · Claude Code","Event-loop order","agent-loop-by-task/Event-loop order","Strict passes per task: single call vs agent loop: Event-loop order",1,"rate","100% (3/3)",3,[0.4385,1],"ci95","Claude Haiku 4.5 · agent loop","\u0001","\u0001","\u0001","higher"],["single-call-vs-agent-loop","agent-loop-by-task","Claude Haiku 4.5 (agent loop) · Claude Code","Room schedule","agent-loop-by-task/Room schedule","Strict passes per task: single call vs agent loop: Room schedule",0.6667,"rate","67% (2/3)",3,[0.2077,0.9385],"ci95","Claude Haiku 4.5 · agent loop","\u0001","\u0001","\u0001","higher"],["single-call-vs-agent-loop","agent-loop-by-task","Claude Haiku 4.5 (agent loop) · Claude Code","SemVer regex","agent-loop-by-task/SemVer regex","Strict passes per task: single call vs agent loop: SemVer regex",0.6667,"rate","67% (2/3)",3,[0.2077,0.9385],"ci95","Claude Haiku 4.5 · agent loop","\u0001","\u0001","\u0001","higher"],["single-call-vs-agent-loop","agent-loop-by-task","Claude Haiku 4.5 (agent loop) · Claude Code","Money refactor","agent-loop-by-task/Money refactor","Strict passes per task: single call vs agent loop: Money refactor",0,"rate","0% (0/3)",3,[0,0.5615],"ci95","Claude Haiku 4.5 · agent loop","\u0001","\u0001","\u0001","higher"],["single-call-vs-agent-loop","agent-loop-by-task","Claude Haiku 4.5 (agent loop) · Claude Code","SQL report","agent-loop-by-task/SQL report","Strict passes per task: single call vs agent loop: SQL report",0,"rate","0% (0/3)",3,[0,0.5615],"ci95","Claude Haiku 4.5 · agent loop","\u0001","\u0001","\u0001","higher"],["single-call-vs-agent-loop","agent-loop-by-task","Claude Sonnet 5.5 (single call) · Claude Code","Interval merge fix","agent-loop-by-task/Interval merge fix","Strict passes per task: single call vs agent loop: Interval merge fix",1,"rate","100% (3/3)",3,[0.4385,1],"ci95","Claude Sonnet 5.5 · single call","\u0001","\u0001","\u0001","higher"],["single-call-vs-agent-loop","agent-loop-by-task","Claude Sonnet 5.5 (single call) · Claude Code","DST day-length fix","agent-loop-by-task/DST day-length fix","Strict passes per task: single call vs agent loop: DST day-length fix",1,"rate","100% (3/3)",3,[0.4385,1],"ci95","Claude Sonnet 5.5 · single call","\u0001","\u0001","\u0001","higher"],["single-call-vs-agent-loop","agent-loop-by-task","Claude Sonnet 5.5 (single call) · Claude Code","CSV parser","agent-loop-by-task/CSV parser","Strict passes per task: single call vs agent loop: CSV parser",1,"rate","100% (3/3)",3,[0.4385,1],"ci95","Claude Sonnet 5.5 · single call","\u0001","\u0001","\u0001","higher"],["single-call-vs-agent-loop","agent-loop-by-task","Claude Sonnet 5.5 (single call) · Claude Code","Event-loop order","agent-loop-by-task/Event-loop order","Strict passes per task: single call vs agent loop: Event-loop order",1,"rate","100% (3/3)",3,[0.4385,1],"ci95","Claude Sonnet 5.5 · single call","\u0001","\u0001","\u0001","higher"],["single-call-vs-agent-loop","agent-loop-by-task","Claude Sonnet 5.5 (single call) · Claude Code","Room schedule","agent-loop-by-task/Room schedule","Strict passes per task: single call vs agent loop: Room schedule",1,"rate","100% (3/3)",3,[0.4385,1],"ci95","Claude Sonnet 5.5 · single call","\u0001","\u0001","\u0001","higher"],["single-call-vs-agent-loop","agent-loop-by-task","Claude Sonnet 5.5 (single call) · Claude Code","SemVer regex","agent-loop-by-task/SemVer regex","Strict passes per task: single call vs agent loop: SemVer regex",1,"rate","100% (3/3)",3,[0.4385,1],"ci95","Claude Sonnet 5.5 · single call","\u0001","\u0001","\u0001","higher"],["single-call-vs-agent-loop","agent-loop-by-task","Claude Sonnet 5.5 (single call) · Claude Code","Money refactor","agent-loop-by-task/Money refactor","Strict passes per task: single call vs agent loop: Money refactor",1,"rate","100% (3/3)",3,[0.4385,1],"ci95","Claude Sonnet 5.5 · single call","\u0001","\u0001","\u0001","higher"],["single-call-vs-agent-loop","agent-loop-by-task","Claude Sonnet 5.5 (single call) · Claude Code","SQL report","agent-loop-by-task/SQL report","Strict passes per task: single call vs agent loop: SQL report",1,"rate","100% (3/3)",3,[0.4385,1],"ci95","Claude Sonnet 5.5 · single call","\u0001","\u0001","\u0001","higher"],["single-call-vs-agent-loop","agent-loop-by-task","Claude Sonnet 5.5 (agent loop) · Claude Code","Interval merge fix","agent-loop-by-task/Interval merge fix","Strict passes per task: single call vs agent loop: Interval merge fix",1,"rate","100% (2/2)",2,[0.3424,1],"ci95","Claude Sonnet 5.5 · agent loop","\u0001","\u0001","\u0001","higher"],["single-call-vs-agent-loop","agent-loop-by-task","Claude Sonnet 5.5 (agent loop) · Claude Code","DST day-length fix","agent-loop-by-task/DST day-length fix","Strict passes per task: single call vs agent loop: DST day-length fix",1,"rate","100% (2/2)",2,[0.3424,1],"ci95","Claude Sonnet 5.5 · agent loop","\u0001","\u0001","\u0001","higher"],["single-call-vs-agent-loop","agent-loop-by-task","Claude Sonnet 5.5 (agent loop) · Claude Code","CSV parser","agent-loop-by-task/CSV parser","Strict passes per task: single call vs agent loop: CSV parser",1,"rate","100% (2/2)",2,[0.3424,1],"ci95","Claude Sonnet 5.5 · agent loop","\u0001","\u0001","\u0001","higher"],["single-call-vs-agent-loop","agent-loop-by-task","Claude Sonnet 5.5 (agent loop) · Claude Code","Event-loop order","agent-loop-by-task/Event-loop order","Strict passes per task: single call vs agent loop: Event-loop order",1,"rate","100% (2/2)",2,[0.3424,1],"ci95","Claude Sonnet 5.5 · agent loop","\u0001","\u0001","\u0001","higher"],["single-call-vs-agent-loop","agent-loop-by-task","Claude Sonnet 5.5 (agent loop) · Claude Code","Room schedule","agent-loop-by-task/Room schedule","Strict passes per task: single call vs agent loop: Room schedule",1,"rate","100% (2/2)",2,[0.3424,1],"ci95","Claude Sonnet 5.5 · agent loop","\u0001","\u0001","\u0001","higher"],["single-call-vs-agent-loop","agent-loop-by-task","Claude Sonnet 5.5 (agent loop) · Claude Code","SemVer regex","agent-loop-by-task/SemVer regex","Strict passes per task: single call vs agent loop: SemVer regex",1,"rate","100% (2/2)",2,[0.3424,1],"ci95","Claude Sonnet 5.5 · agent loop","\u0001","\u0001","\u0001","higher"],["single-call-vs-agent-loop","agent-loop-by-task","Claude Sonnet 5.5 (agent loop) · Claude Code","Money refactor","agent-loop-by-task/Money refactor","Strict passes per task: single call vs agent loop: Money refactor",1,"rate","100% (2/2)",2,[0.3424,1],"ci95","Claude Sonnet 5.5 · agent loop","\u0001","\u0001","\u0001","higher"],["single-call-vs-agent-loop","agent-loop-by-task","Claude Sonnet 5.5 (agent loop) · Claude Code","SQL report","agent-loop-by-task/SQL report","Strict passes per task: single call vs agent loop: SQL report",1,"rate","100% (2/2)",2,[0.3424,1],"ci95","Claude Sonnet 5.5 · agent loop","\u0001","\u0001","\u0001","higher"],["single-call-vs-agent-loop","agent-loop-total-time","Total time per attempt","Claude Haiku 4.5 (single call) · Claude Code","agent-loop-total-time","Total time per attempt: single call vs agent loop",39.01,"seconds","39.0 s",24,"\u0001","minmax","Claude Haiku 4.5 · single call",[15.27,75.13],"\u0001","\u0001","\u0001"],["single-call-vs-agent-loop","agent-loop-total-time","Total time per attempt","Claude Haiku 4.5 (agent loop) · Claude Code","agent-loop-total-time","Total time per attempt: single call vs agent loop",56.77,"seconds","56.8 s",24,"\u0001","minmax","Claude Haiku 4.5 · agent loop",[24.53,223.7],"\u0001","\u0001","\u0001"],["single-call-vs-agent-loop","agent-loop-total-time","Total time per attempt","Claude Sonnet 5.5 (single call) · Claude Code","agent-loop-total-time","Total time per attempt: single call vs agent loop",7.75,"seconds","7.75 s",24,"\u0001","minmax","Claude Sonnet 5.5 · single call",[2.26,34.79],"\u0001","\u0001","\u0001"],["single-call-vs-agent-loop","agent-loop-total-time","Total time per attempt","Claude Sonnet 5.5 (agent loop) · Claude Code","agent-loop-total-time","Total time per attempt: single call vs agent loop",7.41,"seconds","7.41 s",16,"\u0001","minmax","Claude Sonnet 5.5 · agent loop",[2.75,24.19],"\u0001","\u0001","\u0001"],["single-call-vs-agent-loop","agent-loop-tokens","Input tokens (cache reads included)","Claude Haiku 4.5 (single call) · Claude Code","agent-loop-tokens/Input tokens (cache reads included)","Tokens per attempt: single call vs agent loop (Input tokens (cache reads included))",3941,"tokens","3,941",24,"\u0001","minmax","Claude Haiku 4.5 · single call",[3879,4221],"\u0001","\u0001","\u0001"],["single-call-vs-agent-loop","agent-loop-tokens","Input tokens (cache reads included)","Claude Haiku 4.5 (agent loop) · Claude Code","agent-loop-tokens/Input tokens (cache reads included)","Tokens per attempt: single call vs agent loop (Input tokens (cache reads included))",71691,"tokens","71,691",24,"\u0001","minmax","Claude Haiku 4.5 · agent loop",[41732,516306],"\u0001","\u0001","\u0001"],["single-call-vs-agent-loop","agent-loop-tokens","Input tokens (cache reads included)","Claude Sonnet 5.5 (single call) · Claude Code","agent-loop-tokens/Input tokens (cache reads included)","Tokens per attempt: single call vs agent loop (Input tokens (cache reads included))",2281,"tokens","2,281",24,"\u0001","minmax","Claude Sonnet 5.5 · single call",[2234,2669],"\u0001","\u0001","\u0001"],["single-call-vs-agent-loop","agent-loop-tokens","Input tokens (cache reads included)","Claude Sonnet 5.5 (agent loop) · Claude Code","agent-loop-tokens/Input tokens (cache reads included)","Tokens per attempt: single call vs agent loop (Input tokens (cache reads included))",9550,"tokens","9,550",16,"\u0001","minmax","Claude Sonnet 5.5 · agent loop",[9398,33040],"\u0001","\u0001","\u0001"],["single-call-vs-agent-loop","agent-loop-tokens","Output tokens","Claude Haiku 4.5 (single call) · Claude Code","agent-loop-tokens/Output tokens","Tokens per attempt: single call vs agent loop (Output tokens)",5064,"tokens","5,064",24,"\u0001","minmax","Claude Haiku 4.5 · single call",[1899,9321],"\u0001","\u0001","\u0001"],["single-call-vs-agent-loop","agent-loop-tokens","Output tokens","Claude Haiku 4.5 (agent loop) · Claude Code","agent-loop-tokens/Output tokens","Tokens per attempt: single call vs agent loop (Output tokens)",7912,"tokens","7,912",24,"\u0001","minmax","Claude Haiku 4.5 · agent loop",[2541,20654],"\u0001","\u0001","\u0001"],["single-call-vs-agent-loop","agent-loop-tokens","Output tokens","Claude Sonnet 5.5 (single call) · Claude Code","agent-loop-tokens/Output tokens","Tokens per attempt: single call vs agent loop (Output tokens)",1050,"tokens","1,050",24,"\u0001","minmax","Claude Sonnet 5.5 · single call",[176,3895],"\u0001","\u0001","\u0001"],["single-call-vs-agent-loop","agent-loop-tokens","Output tokens","Claude Sonnet 5.5 (agent loop) · Claude Code","agent-loop-tokens/Output tokens","Tokens per attempt: single call vs agent loop (Output tokens)",876,"tokens","876",16,"\u0001","minmax","Claude Sonnet 5.5 · agent loop",[219,3243],"\u0001","\u0001","\u0001"],["single-call-vs-agent-loop","agent-loop-tool-calls","Tool calls per attempt","Claude Haiku 4.5 (agent loop) · Claude Code","agent-loop-tool-calls","Tool calls per agent-loop attempt",3,"count","3",24,"\u0001","minmax","Claude Haiku 4.5 · agent loop",[2,18],"\u0001","\u0001","\u0001"],["single-call-vs-agent-loop","agent-loop-tool-calls","Tool calls per attempt","Claude Sonnet 5.5 (agent loop) · Claude Code","agent-loop-tool-calls","Tool calls per agent-loop attempt",0,"count","0",16,"\u0001","minmax","Claude Sonnet 5.5 · agent loop",[0,3],"\u0001","\u0001","\u0001"],["single-call-vs-agent-loop","agent-loop-cost-per-pass","Cost per strict pass","Claude Haiku 4.5 (single call) · Claude Code","agent-loop-cost-per-pass","List-price cost per strict pass: single call vs agent loop (calculation)",0.0672,"usd","$0.067",24,"\u0001","\u0001","Claude Haiku 4.5 · single call","\u0001",true,"\u0001","\u0001"],["single-call-vs-agent-loop","agent-loop-cost-per-pass","Cost per strict pass","Claude Haiku 4.5 (agent loop) · Claude Code","agent-loop-cost-per-pass","List-price cost per strict pass: single call vs agent loop (calculation)",0.14225,"usd","$0.14",24,"\u0001","\u0001","Claude Haiku 4.5 · agent loop","\u0001",true,"\u0001","\u0001"],["single-call-vs-agent-loop","agent-loop-cost-per-pass","Cost per strict pass","Claude Sonnet 5.5 (single call) · Claude Code","agent-loop-cost-per-pass","List-price cost per strict pass: single call vs agent loop (calculation)",0.01435,"usd","$0.014",24,"\u0001","\u0001","Claude Sonnet 5.5 · single call","\u0001",true,"\u0001","\u0001"],["single-call-vs-agent-loop","agent-loop-cost-per-pass","Cost per strict pass","Claude Sonnet 5.5 (agent loop) · Claude Code","agent-loop-cost-per-pass","List-price cost per strict pass: single call vs agent loop (calculation)",0.02746,"usd","$0.027",16,"\u0001","\u0001","Claude Sonnet 5.5 · agent loop","\u0001",true,"\u0001","\u0001"],["haiku-thinking-on-off","haiku-thinking-router-exact","Exact decisions (every scored question right)","Claude Haiku 4.5 (thinking off) · Claude Code","haiku-thinking-router-exact/Exact decisions (every scored question right)","Haiku thinking study: typed routing decisions answered exactly right (Exact decisions (every scored question right))",0.8659,"rate","87% (71/82)",82,[0.7755,0.9234],"ci95","Claude Haiku 4.5 · thinking off · typed routing decisions, thinking on vs off","\u0001","\u0001","\u0001","higher"],["haiku-thinking-on-off","haiku-thinking-router-exact","Exact decisions (every scored question right)","Claude Haiku 4.5 (thinking on) · Claude Code","haiku-thinking-router-exact/Exact decisions (every scored question right)","Haiku thinking study: typed routing decisions answered exactly right (Exact decisions (every scored question right))",0.8902,"rate","89% (73/82)",82,[0.8044,0.9412],"ci95","Claude Haiku 4.5 · thinking on · typed routing decisions, thinking on vs off","\u0001","\u0001","\u0001","higher"],["haiku-thinking-on-off","haiku-thinking-router-exact","Exact decisions (every scored question right)","Claude Sonnet 5.5 (low) · Claude Code","haiku-thinking-router-exact/Exact decisions (every scored question right)","Haiku thinking study: typed routing decisions answered exactly right (Exact decisions (every scored question right))",0.939,"rate","94% (77/82)",82,[0.8651,0.9737],"ci95","Claude Sonnet 5.5 · effort low · typed routing decisions, thinking on vs off","\u0001","\u0001","\u0001","higher"],["haiku-thinking-on-off","haiku-thinking-router-exact","Per-question accuracy","Claude Haiku 4.5 (thinking off) · Claude Code","haiku-thinking-router-exact/Per-question accuracy","Haiku thinking study: typed routing decisions answered exactly right (Per-question accuracy)",0.9124,"rate","91% (177/194)",194,[0.8642,0.9446],"ci95","Claude Haiku 4.5 · thinking off · typed routing decisions, thinking on vs off","\u0001","\u0001","\u0001","higher"],["haiku-thinking-on-off","haiku-thinking-router-exact","Per-question accuracy","Claude Haiku 4.5 (thinking on) · Claude Code","haiku-thinking-router-exact/Per-question accuracy","Haiku thinking study: typed routing decisions answered exactly right (Per-question accuracy)",0.9433,"rate","94% (183/194)",194,[0.9013,0.968],"ci95","Claude Haiku 4.5 · thinking on · typed routing decisions, thinking on vs off","\u0001","\u0001","\u0001","higher"],["haiku-thinking-on-off","haiku-thinking-router-exact","Per-question accuracy","Claude Sonnet 5.5 (low) · Claude Code","haiku-thinking-router-exact/Per-question accuracy","Haiku thinking study: typed routing decisions answered exactly right (Per-question accuracy)",0.9742,"rate","97% (189/194)",194,[0.9411,0.9889],"ci95","Claude Sonnet 5.5 · effort low · typed routing decisions, thinking on vs off","\u0001","\u0001","\u0001","higher"],["haiku-thinking-on-off","haiku-thinking-router-latency","Wall time (CLI)","Claude Haiku 4.5 (thinking off) · Claude Code","haiku-thinking-router-latency/Wall time (CLI)","Haiku thinking study: time per routing decision (Wall time (CLI))",4.66,"seconds","4.66 s",82,"\u0001","p50-p95","Claude Haiku 4.5 · thinking off · typed routing decisions, thinking on vs off",[4.66,8.18],"\u0001","\u0001","\u0001"],["haiku-thinking-on-off","haiku-thinking-router-latency","Wall time (CLI)","Claude Haiku 4.5 (thinking on) · Claude Code","haiku-thinking-router-latency/Wall time (CLI)","Haiku thinking study: time per routing decision (Wall time (CLI))",12.54,"seconds","12.5 s",82,"\u0001","p50-p95","Claude Haiku 4.5 · thinking on · typed routing decisions, thinking on vs off",[12.54,34.48],"\u0001","\u0001","\u0001"],["haiku-thinking-on-off","haiku-thinking-router-latency","Wall time (CLI)","Claude Sonnet 5.5 (low) · Claude Code","haiku-thinking-router-latency/Wall time (CLI)","Haiku thinking study: time per routing decision (Wall time (CLI))",2.6,"seconds","2.60 s",82,"\u0001","p50-p95","Claude Sonnet 5.5 · effort low · typed routing decisions, thinking on vs off",[2.6,4.3],"\u0001","\u0001","\u0001"],["haiku-thinking-on-off","haiku-thinking-router-latency","Model time (API)","Claude Haiku 4.5 (thinking off) · Claude Code","haiku-thinking-router-latency/Model time (API)","Haiku thinking study: time per routing decision (Model time (API))",3.79,"seconds","3.79 s",82,"\u0001","p50-p95","Claude Haiku 4.5 · thinking off · typed routing decisions, thinking on vs off",[3.79,7.43],"\u0001","\u0001","\u0001"],["haiku-thinking-on-off","haiku-thinking-router-latency","Model time (API)","Claude Haiku 4.5 (thinking on) · Claude Code","haiku-thinking-router-latency/Model time (API)","Haiku thinking study: time per routing decision (Model time (API))",10.51,"seconds","10.5 s",82,"\u0001","p50-p95","Claude Haiku 4.5 · thinking on · typed routing decisions, thinking on vs off",[10.51,32.13],"\u0001","\u0001","\u0001"],["haiku-thinking-on-off","haiku-thinking-router-latency","Model time (API)","Claude Sonnet 5.5 (low) · Claude Code","haiku-thinking-router-latency/Model time (API)","Haiku thinking study: time per routing decision (Model time (API))",1.6,"seconds","1.60 s",82,"\u0001","p50-p95","Claude Sonnet 5.5 · effort low · typed routing decisions, thinking on vs off",[1.6,2.58],"\u0001","\u0001","\u0001"],["haiku-thinking-on-off","haiku-thinking-router-tokens","Thinking tokens","Claude Haiku 4.5 (thinking off) · Claude Code","haiku-thinking-router-tokens/Thinking tokens","Haiku thinking study: thinking and visible output tokens per routing decision (Thinking tokens)",0,"tokens","0",82,"\u0001","\u0001","Claude Haiku 4.5 · thinking off · typed routing decisions, thinking on vs off","\u0001","\u0001","\u0001","\u0001"],["haiku-thinking-on-off","haiku-thinking-router-tokens","Thinking tokens","Claude Haiku 4.5 (thinking on) · Claude Code","haiku-thinking-router-tokens/Thinking tokens","Haiku thinking study: thinking and visible output tokens per routing decision (Thinking tokens)",1101,"tokens","1,101",82,"\u0001","\u0001","Claude Haiku 4.5 · thinking on · typed routing decisions, thinking on vs off","\u0001","\u0001","\u0001","\u0001"],["haiku-thinking-on-off","haiku-thinking-router-tokens","Thinking tokens","Claude Sonnet 5.5 (low) · Claude Code","haiku-thinking-router-tokens/Thinking tokens","Haiku thinking study: thinking and visible output tokens per routing decision (Thinking tokens)",2,"tokens","2",82,"\u0001","\u0001","Claude Sonnet 5.5 · effort low · typed routing decisions, thinking on vs off","\u0001","\u0001","\u0001","\u0001"],["haiku-thinking-on-off","haiku-thinking-router-tokens","Visible output tokens","Claude Haiku 4.5 (thinking off) · Claude Code","haiku-thinking-router-tokens/Visible output tokens","Haiku thinking study: thinking and visible output tokens per routing decision (Visible output tokens)",366,"tokens","366",82,"\u0001","\u0001","Claude Haiku 4.5 · thinking off · typed routing decisions, thinking on vs off","\u0001","\u0001","\u0001","\u0001"],["haiku-thinking-on-off","haiku-thinking-router-tokens","Visible output tokens","Claude Haiku 4.5 (thinking on) · Claude Code","haiku-thinking-router-tokens/Visible output tokens","Haiku thinking study: thinking and visible output tokens per routing decision (Visible output tokens)",318,"tokens","318",82,"\u0001","\u0001","Claude Haiku 4.5 · thinking on · typed routing decisions, thinking on vs off","\u0001","\u0001","\u0001","\u0001"],["haiku-thinking-on-off","haiku-thinking-router-tokens","Visible output tokens","Claude Sonnet 5.5 (low) · Claude Code","haiku-thinking-router-tokens/Visible output tokens","Haiku thinking study: thinking and visible output tokens per routing decision (Visible output tokens)",105,"tokens","105",82,"\u0001","\u0001","Claude Sonnet 5.5 · effort low · typed routing decisions, thinking on vs off","\u0001","\u0001","\u0001","\u0001"],["haiku-thinking-on-off","haiku-thinking-router-cost","Cost per 1,000 decisions","Claude Haiku 4.5 (thinking off) · Claude Code","haiku-thinking-router-cost","Haiku thinking study: list-price cost per 1,000 routing decisions (calculation)",3.364,"usd","$3.36",82,"\u0001","\u0001","Claude Haiku 4.5 · thinking off · typed routing decisions, thinking on vs off","\u0001",true,"\u0001","\u0001"],["haiku-thinking-on-off","haiku-thinking-router-cost","Cost per 1,000 decisions","Claude Haiku 4.5 (thinking on) · Claude Code","haiku-thinking-router-cost","Haiku thinking study: list-price cost per 1,000 routing decisions (calculation)",8.924,"usd","$8.92",82,"\u0001","\u0001","Claude Haiku 4.5 · thinking on · typed routing decisions, thinking on vs off","\u0001",true,"\u0001","\u0001"],["haiku-thinking-on-off","haiku-thinking-router-cost","Cost per 1,000 decisions","Claude Sonnet 5.5 (low) · Claude Code","haiku-thinking-router-cost","Haiku thinking study: list-price cost per 1,000 routing decisions (calculation)",7.324,"usd","$7.32",82,"\u0001","\u0001","Claude Sonnet 5.5 · effort low · typed routing decisions, thinking on vs off","\u0001",true,"\u0001","\u0001"],["haiku-thinking-on-off","haiku-thinking-hard-pass","Strict pass","Claude Haiku 4.5 (thinking off) · Claude Code","haiku-thinking-hard-pass/Strict pass","Haiku thinking study: pass rate on eight hard tasks (Strict pass)",0.1667,"rate","17% (4/24)",24,[0.0668,0.3585],"ci95","Claude Haiku 4.5 · thinking off · eight hard validated tasks, thinking on vs off","\u0001","\u0001","\u0001","higher"],["haiku-thinking-on-off","haiku-thinking-hard-pass","Strict pass","Claude Haiku 4.5 (thinking on) · Claude Code","haiku-thinking-hard-pass/Strict pass","Haiku thinking study: pass rate on eight hard tasks (Strict pass)",0.4583,"rate","46% (11/24)",24,[0.2789,0.6493],"ci95","Claude Haiku 4.5 · thinking on · eight hard validated tasks, thinking on vs off","\u0001","\u0001","\u0001","higher"],["haiku-thinking-on-off","haiku-thinking-hard-pass","Lenient (format misses counted)","Claude Haiku 4.5 (thinking off) · Claude Code","haiku-thinking-hard-pass/Lenient (format misses counted)","Haiku thinking study: pass rate on eight hard tasks (Lenient (format misses counted))",0.1667,"rate","17% (4/24)",24,[0.0668,0.3585],"ci95","Claude Haiku 4.5 · thinking off · eight hard validated tasks, thinking on vs off","\u0001","\u0001","\u0001","higher"],["haiku-thinking-on-off","haiku-thinking-hard-pass","Lenient (format misses counted)","Claude Haiku 4.5 (thinking on) · Claude Code","haiku-thinking-hard-pass/Lenient (format misses counted)","Haiku thinking study: pass rate on eight hard tasks (Lenient (format misses counted))",0.6667,"rate","67% (16/24)",24,[0.4671,0.8203],"ci95","Claude Haiku 4.5 · thinking on · eight hard validated tasks, thinking on vs off","\u0001","\u0001","\u0001","higher"],["haiku-thinking-on-off","haiku-thinking-hard-time","Total time per call","Claude Haiku 4.5 (thinking off) · Claude Code","haiku-thinking-hard-time","Haiku thinking study: total time per call on hard tasks",2.95,"seconds","2.95 s",24,"\u0001","minmax","Claude Haiku 4.5 · thinking off · eight hard validated tasks, thinking on vs off",[1.7,13.01],"\u0001","\u0001","\u0001"],["haiku-thinking-on-off","haiku-thinking-hard-time","Total time per call","Claude Haiku 4.5 (thinking on) · Claude Code","haiku-thinking-hard-time","Haiku thinking study: total time per call on hard tasks",39.01,"seconds","39.0 s",24,"\u0001","minmax","Claude Haiku 4.5 · thinking on · eight hard validated tasks, thinking on vs off",[15.27,75.13],"\u0001","\u0001","\u0001"],["json-schema-vs-instructions","structured-output-pass-rate","Strict pass: the whole reply is the right JSON","Claude Haiku 4.5 (instructions) · Claude Code","structured-output-pass-rate/Strict pass: the whole reply is the right JSON","Does a JSON schema raise the pass rate? Instructions vs schema mode (Strict pass: the whole reply is the right JSON)",0,"rate","0% (0/24)",24,[0,0.138],"ci95","Claude Haiku 4.5 · instructions","\u0001",true,"\u0001","higher"],["json-schema-vs-instructions","structured-output-pass-rate","Strict pass: the whole reply is the right JSON","Claude Haiku 4.5 (JSON schema) · Claude Code","structured-output-pass-rate/Strict pass: the whole reply is the right JSON","Does a JSON schema raise the pass rate? Instructions vs schema mode (Strict pass: the whole reply is the right JSON)",0.75,"rate","75% (18/24)",24,[0.551,0.88],"ci95","Claude Haiku 4.5 · JSON schema","\u0001",true,"\u0001","higher"],["json-schema-vs-instructions","structured-output-pass-rate","Strict pass: the whole reply is the right JSON","Claude Sonnet 5.5 (instructions) · Claude Code","structured-output-pass-rate/Strict pass: the whole reply is the right JSON","Does a JSON schema raise the pass rate? Instructions vs schema mode (Strict pass: the whole reply is the right JSON)",1,"rate","100% (12/12)",12,[0.7575,1],"ci95","Claude Sonnet 5.5 · instructions","\u0001",true,"\u0001","higher"],["json-schema-vs-instructions","structured-output-pass-rate","Strict pass: the whole reply is the right JSON","Claude Sonnet 5.5 (JSON schema) · Claude Code","structured-output-pass-rate/Strict pass: the whole reply is the right JSON","Does a JSON schema raise the pass rate? Instructions vs schema mode (Strict pass: the whole reply is the right JSON)",1,"rate","100% (12/12)",12,[0.7575,1],"ci95","Claude Sonnet 5.5 · JSON schema","\u0001",true,"\u0001","higher"],["json-schema-vs-instructions","structured-output-pass-rate","Right answer in any format (strict pass or format miss)","Claude Haiku 4.5 (instructions) · Claude Code","structured-output-pass-rate/Right answer in any format (strict pass or format miss)","Does a JSON schema raise the pass rate? Instructions vs schema mode (Right answer in any format (strict pass or format miss))",0.7083,"rate","71% (17/24)",24,[0.5083,0.8509],"ci95","Claude Haiku 4.5 · instructions","\u0001",true,"\u0001","higher"],["json-schema-vs-instructions","structured-output-pass-rate","Right answer in any format (strict pass or format miss)","Claude Haiku 4.5 (JSON schema) · Claude Code","structured-output-pass-rate/Right answer in any format (strict pass or format miss)","Does a JSON schema raise the pass rate? Instructions vs schema mode (Right answer in any format (strict pass or format miss))",0.75,"rate","75% (18/24)",24,[0.551,0.88],"ci95","Claude Haiku 4.5 · JSON schema","\u0001",true,"\u0001","higher"],["json-schema-vs-instructions","structured-output-pass-rate","Right answer in any format (strict pass or format miss)","Claude Sonnet 5.5 (instructions) · Claude Code","structured-output-pass-rate/Right answer in any format (strict pass or format miss)","Does a JSON schema raise the pass rate? Instructions vs schema mode (Right answer in any format (strict pass or format miss))",1,"rate","100% (12/12)",12,[0.7575,1],"ci95","Claude Sonnet 5.5 · instructions","\u0001",true,"\u0001","higher"],["json-schema-vs-instructions","structured-output-pass-rate","Right answer in any format (strict pass or format miss)","Claude Sonnet 5.5 (JSON schema) · Claude Code","structured-output-pass-rate/Right answer in any format (strict pass or format miss)","Does a JSON schema raise the pass rate? Instructions vs schema mode (Right answer in any format (strict pass or format miss))",1,"rate","100% (12/12)",12,[0.7575,1],"ci95","Claude Sonnet 5.5 · JSON schema","\u0001",true,"\u0001","higher"],["json-schema-vs-instructions","structured-output-outcomes","Strict pass","Claude Haiku 4.5 (instructions) · Claude Code","structured-output-outcomes/Strict pass","What each call produced: strict pass, format miss, wrong values or error (Strict pass)",0,"count","0",24,"\u0001","\u0001","Claude Haiku 4.5 · instructions","\u0001","\u0001","\u0001","\u0001"],["json-schema-vs-instructions","structured-output-outcomes","Strict pass","Claude Haiku 4.5 (JSON schema) · Claude Code","structured-output-outcomes/Strict pass","What each call produced: strict pass, format miss, wrong values or error (Strict pass)",18,"count","18",24,"\u0001","\u0001","Claude Haiku 4.5 · JSON schema","\u0001","\u0001","\u0001","\u0001"],["json-schema-vs-instructions","structured-output-outcomes","Strict pass","Claude Sonnet 5.5 (instructions) · Claude Code","structured-output-outcomes/Strict pass","What each call produced: strict pass, format miss, wrong values or error (Strict pass)",12,"count","12",12,"\u0001","\u0001","Claude Sonnet 5.5 · instructions","\u0001","\u0001","\u0001","\u0001"],["json-schema-vs-instructions","structured-output-outcomes","Strict pass","Claude Sonnet 5.5 (JSON schema) · Claude Code","structured-output-outcomes/Strict pass","What each call produced: strict pass, format miss, wrong values or error (Strict pass)",12,"count","12",12,"\u0001","\u0001","Claude Sonnet 5.5 · JSON schema","\u0001","\u0001","\u0001","\u0001"],["json-schema-vs-instructions","structured-output-outcomes","Format miss","Claude Haiku 4.5 (instructions) · Claude Code","structured-output-outcomes/Format miss","What each call produced: strict pass, format miss, wrong values or error (Format miss)",17,"count","17",24,"\u0001","\u0001","Claude Haiku 4.5 · instructions","\u0001","\u0001","\u0001","\u0001"],["json-schema-vs-instructions","structured-output-outcomes","Format miss","Claude Haiku 4.5 (JSON schema) · Claude Code","structured-output-outcomes/Format miss","What each call produced: strict pass, format miss, wrong values or error (Format miss)",0,"count","0",24,"\u0001","\u0001","Claude Haiku 4.5 · JSON schema","\u0001","\u0001","\u0001","\u0001"],["json-schema-vs-instructions","structured-output-outcomes","Format miss","Claude Sonnet 5.5 (instructions) · Claude Code","structured-output-outcomes/Format miss","What each call produced: strict pass, format miss, wrong values or error (Format miss)",0,"count","0",12,"\u0001","\u0001","Claude Sonnet 5.5 · instructions","\u0001","\u0001","\u0001","\u0001"],["json-schema-vs-instructions","structured-output-outcomes","Format miss","Claude Sonnet 5.5 (JSON schema) · Claude Code","structured-output-outcomes/Format miss","What each call produced: strict pass, format miss, wrong values or error (Format miss)",0,"count","0",12,"\u0001","\u0001","Claude Sonnet 5.5 · JSON schema","\u0001","\u0001","\u0001","\u0001"],["json-schema-vs-instructions","structured-output-outcomes","Wrong values","Claude Haiku 4.5 (instructions) · Claude Code","structured-output-outcomes/Wrong values","What each call produced: strict pass, format miss, wrong values or error (Wrong values)",7,"count","7",24,"\u0001","\u0001","Claude Haiku 4.5 · instructions","\u0001","\u0001","\u0001","\u0001"],["json-schema-vs-instructions","structured-output-outcomes","Wrong values","Claude Haiku 4.5 (JSON schema) · Claude Code","structured-output-outcomes/Wrong values","What each call produced: strict pass, format miss, wrong values or error (Wrong values)",6,"count","6",24,"\u0001","\u0001","Claude Haiku 4.5 · JSON schema","\u0001","\u0001","\u0001","\u0001"],["json-schema-vs-instructions","structured-output-outcomes","Wrong values","Claude Sonnet 5.5 (instructions) · Claude Code","structured-output-outcomes/Wrong values","What each call produced: strict pass, format miss, wrong values or error (Wrong values)",0,"count","0",12,"\u0001","\u0001","Claude Sonnet 5.5 · instructions","\u0001","\u0001","\u0001","\u0001"],["json-schema-vs-instructions","structured-output-outcomes","Wrong values","Claude Sonnet 5.5 (JSON schema) · Claude Code","structured-output-outcomes/Wrong values","What each call produced: strict pass, format miss, wrong values or error (Wrong values)",0,"count","0",12,"\u0001","\u0001","Claude Sonnet 5.5 · JSON schema","\u0001","\u0001","\u0001","\u0001"],["json-schema-vs-instructions","structured-output-outcomes","Error","Claude Haiku 4.5 (instructions) · Claude Code","structured-output-outcomes/Error","What each call produced: strict pass, format miss, wrong values or error (Error)",0,"count","0",24,"\u0001","\u0001","Claude Haiku 4.5 · instructions","\u0001","\u0001","\u0001","\u0001"],["json-schema-vs-instructions","structured-output-outcomes","Error","Claude Haiku 4.5 (JSON schema) · Claude Code","structured-output-outcomes/Error","What each call produced: strict pass, format miss, wrong values or error (Error)",0,"count","0",24,"\u0001","\u0001","Claude Haiku 4.5 · JSON schema","\u0001","\u0001","\u0001","\u0001"],["json-schema-vs-instructions","structured-output-outcomes","Error","Claude Sonnet 5.5 (instructions) · Claude Code","structured-output-outcomes/Error","What each call produced: strict pass, format miss, wrong values or error (Error)",0,"count","0",12,"\u0001","\u0001","Claude Sonnet 5.5 · instructions","\u0001","\u0001","\u0001","\u0001"],["json-schema-vs-instructions","structured-output-outcomes","Error","Claude Sonnet 5.5 (JSON schema) · Claude Code","structured-output-outcomes/Error","What each call produced: strict pass, format miss, wrong values or error (Error)",0,"count","0",12,"\u0001","\u0001","Claude Sonnet 5.5 · JSON schema","\u0001","\u0001","\u0001","\u0001"],["json-schema-vs-instructions","structured-output-time","Median time per call (the three prompts pooled)","Claude Haiku 4.5 (instructions) · Claude Code","structured-output-time","Time per call, instructions vs schema mode",9.52,"seconds","9.52 s",24,"\u0001","minmax","Claude Haiku 4.5 · instructions",[5.67,17],"\u0001","\u0001","\u0001"],["json-schema-vs-instructions","structured-output-time","Median time per call (the three prompts pooled)","Claude Haiku 4.5 (JSON schema) · Claude Code","structured-output-time","Time per call, instructions vs schema mode",8.46,"seconds","8.46 s",24,"\u0001","minmax","Claude Haiku 4.5 · JSON schema",[5.9,12.23],"\u0001","\u0001","\u0001"],["json-schema-vs-instructions","structured-output-time","Median time per call (the three prompts pooled)","Claude Sonnet 5.5 (instructions) · Claude Code","structured-output-time","Time per call, instructions vs schema mode",3.52,"seconds","3.52 s",12,"\u0001","minmax","Claude Sonnet 5.5 · instructions",[2.67,4.12],"\u0001","\u0001","\u0001"],["json-schema-vs-instructions","structured-output-time","Median time per call (the three prompts pooled)","Claude Sonnet 5.5 (JSON schema) · Claude Code","structured-output-time","Time per call, instructions vs schema mode",4.2,"seconds","4.20 s",12,"\u0001","minmax","Claude Sonnet 5.5 · JSON schema",[2.95,6.14],"\u0001","\u0001","\u0001"],["json-schema-vs-instructions","structured-output-tokens","Median output tokens per call","Claude Haiku 4.5 (instructions) · Claude Code","structured-output-tokens/Median output tokens per call","Output and reasoning tokens per call, instructions vs schema mode (Median output tokens per call)",1128,"tokens","1,128",24,"\u0001","\u0001","Claude Haiku 4.5 · instructions","\u0001","\u0001","\u0001","\u0001"],["json-schema-vs-instructions","structured-output-tokens","Median output tokens per call","Claude Haiku 4.5 (JSON schema) · Claude Code","structured-output-tokens/Median output tokens per call","Output and reasoning tokens per call, instructions vs schema mode (Median output tokens per call)",1036,"tokens","1,036",24,"\u0001","\u0001","Claude Haiku 4.5 · JSON schema","\u0001","\u0001","\u0001","\u0001"],["json-schema-vs-instructions","structured-output-tokens","Median output tokens per call","Claude Sonnet 5.5 (instructions) · Claude Code","structured-output-tokens/Median output tokens per call","Output and reasoning tokens per call, instructions vs schema mode (Median output tokens per call)",368,"tokens","368",12,"\u0001","\u0001","Claude Sonnet 5.5 · instructions","\u0001","\u0001","\u0001","\u0001"],["json-schema-vs-instructions","structured-output-tokens","Median output tokens per call","Claude Sonnet 5.5 (JSON schema) · Claude Code","structured-output-tokens/Median output tokens per call","Output and reasoning tokens per call, instructions vs schema mode (Median output tokens per call)",424,"tokens","424",12,"\u0001","\u0001","Claude Sonnet 5.5 · JSON schema","\u0001","\u0001","\u0001","\u0001"],["routing-holdout","routing-holdout-exact","Exact rate","Claude Haiku 4.5 · Claude Code","routing-holdout-exact","Unseen routing decisions answered exactly right",0.7857,"rate","79% (44/56)",56,[0.6618,0.8729],"ci95","Claude Haiku 4.5","\u0001","\u0001","\u0001","higher"],["routing-holdout","routing-holdout-exact","Exact rate","Claude Sonnet 5.5 (low) · Claude Code","routing-holdout-exact","Unseen routing decisions answered exactly right",0.875,"rate","88% (49/56)",56,[0.7637,0.9381],"ci95","Claude Sonnet 5.5 · effort low","\u0001","\u0001","\u0001","higher"],["routing-holdout","routing-holdout-key-accuracy","Key accuracy","Claude Haiku 4.5 · Claude Code","routing-holdout-key-accuracy","Per-question accuracy on unseen decisions",0.816,"rate","82% (102/125)",125,[0.739,0.8741],"ci95","Claude Haiku 4.5","\u0001","\u0001","\u0001","higher"],["routing-holdout","routing-holdout-key-accuracy","Key accuracy","Claude Sonnet 5.5 (low) · Claude Code","routing-holdout-key-accuracy","Per-question accuracy on unseen decisions",0.92,"rate","92% (115/125)",125,[0.859,0.956],"ci95","Claude Sonnet 5.5 · effort low","\u0001","\u0001","\u0001","higher"],["routing-holdout","routing-holdout-by-purpose","Claude Haiku 4.5 · Claude Code","Failure class","routing-holdout-by-purpose/Failure class","Exact rate on unseen decisions, by decision type: Failure class",0.9286,"rate","93% (13/14)",14,[0.6853,0.9873],"ci95","Claude Haiku 4.5","\u0001","\u0001","\u0001","higher"],["routing-holdout","routing-holdout-by-purpose","Claude Haiku 4.5 · Claude Code","Message intent","routing-holdout-by-purpose/Message intent","Exact rate on unseen decisions, by decision type: Message intent",0.9286,"rate","93% (13/14)",14,[0.6853,0.9873],"ci95","Claude Haiku 4.5","\u0001","\u0001","\u0001","higher"],["routing-holdout","routing-holdout-by-purpose","Claude Haiku 4.5 · Claude Code","Is it a rule?","routing-holdout-by-purpose/Is it a rule?","Exact rate on unseen decisions, by decision type: Is it a rule?",0.9286,"rate","93% (13/14)",14,[0.6853,0.9873],"ci95","Claude Haiku 4.5","\u0001","\u0001","\u0001","higher"],["routing-holdout","routing-holdout-by-purpose","Claude Haiku 4.5 · Claude Code","Context shape","routing-holdout-by-purpose/Context shape","Exact rate on unseen decisions, by decision type: Context shape",0.3571,"rate","36% (5/14)",14,[0.1634,0.6124],"ci95","Claude Haiku 4.5","\u0001","\u0001","\u0001","higher"],["routing-holdout","routing-holdout-by-purpose","Claude Sonnet 5.5 (low) · Claude Code","Failure class","routing-holdout-by-purpose/Failure class","Exact rate on unseen decisions, by decision type: Failure class",1,"rate","100% (14/14)",14,[0.7847,1],"ci95","Claude Sonnet 5.5 · effort low","\u0001","\u0001","\u0001","higher"],["routing-holdout","routing-holdout-by-purpose","Claude Sonnet 5.5 (low) · Claude Code","Message intent","routing-holdout-by-purpose/Message intent","Exact rate on unseen decisions, by decision type: Message intent",1,"rate","100% (14/14)",14,[0.7847,1],"ci95","Claude Sonnet 5.5 · effort low","\u0001","\u0001","\u0001","higher"],["routing-holdout","routing-holdout-by-purpose","Claude Sonnet 5.5 (low) · Claude Code","Is it a rule?","routing-holdout-by-purpose/Is it a rule?","Exact rate on unseen decisions, by decision type: Is it a rule?",0.9286,"rate","93% (13/14)",14,[0.6853,0.9873],"ci95","Claude Sonnet 5.5 · effort low","\u0001","\u0001","\u0001","higher"],["routing-holdout","routing-holdout-by-purpose","Claude Sonnet 5.5 (low) · Claude Code","Context shape","routing-holdout-by-purpose/Context shape","Exact rate on unseen decisions, by decision type: Context shape",0.5714,"rate","57% (8/14)",14,[0.3259,0.7862],"ci95","Claude Sonnet 5.5 · effort low","\u0001","\u0001","\u0001","higher"],["routing-holdout","routing-holdout-tuned-vs-unseen","Tuned set (routing-jev-vs-llm)","Claude Haiku 4.5 · Claude Code","routing-holdout-tuned-vs-unseen/Tuned set (routing-jev-vs-llm)","Tuned case set vs unseen holdout: exact rate per router (Tuned set (routing-jev-vs-llm))",0.8902,"rate","89% (73/82)",82,[0.8044,0.9412],"ci95","Claude Haiku 4.5","\u0001","\u0001","\u0001","higher"],["routing-holdout","routing-holdout-tuned-vs-unseen","Tuned set (routing-jev-vs-llm)","Claude Sonnet 5.5 (low) · Claude Code","routing-holdout-tuned-vs-unseen/Tuned set (routing-jev-vs-llm)","Tuned case set vs unseen holdout: exact rate per router (Tuned set (routing-jev-vs-llm))",0.939,"rate","94% (77/82)",82,[0.8651,0.9737],"ci95","Claude Sonnet 5.5 · effort low","\u0001","\u0001","\u0001","higher"],["routing-holdout","routing-holdout-tuned-vs-unseen","Unseen holdout","Claude Haiku 4.5 · Claude Code","routing-holdout-tuned-vs-unseen/Unseen holdout","Tuned case set vs unseen holdout: exact rate per router (Unseen holdout)",0.7857,"rate","79% (44/56)",56,[0.6618,0.8729],"ci95","Claude Haiku 4.5","\u0001","\u0001","\u0001","higher"],["routing-holdout","routing-holdout-tuned-vs-unseen","Unseen holdout","Claude Sonnet 5.5 (low) · Claude Code","routing-holdout-tuned-vs-unseen/Unseen holdout","Tuned case set vs unseen holdout: exact rate per router (Unseen holdout)",0.875,"rate","88% (49/56)",56,[0.7637,0.9381],"ci95","Claude Sonnet 5.5 · effort low","\u0001","\u0001","\u0001","higher"],["routing-holdout","routing-holdout-latency","Wall time","Claude Haiku 4.5 · Claude Code","routing-holdout-latency/Wall time","Time per routing decision, by route (Wall time)",9.444,"seconds","9.44 s",56,"\u0001","p50-p95","Claude Haiku 4.5",[9.444,25.413],"\u0001","\u0001","\u0001"],["routing-holdout","routing-holdout-latency","Wall time","Claude Sonnet 5.5 (low) · Claude Code","routing-holdout-latency/Wall time","Time per routing decision, by route (Wall time)",2.359,"seconds","2.36 s",56,"\u0001","p50-p95","Claude Sonnet 5.5 · effort low",[2.359,3.657],"\u0001","\u0001","\u0001"],["routing-holdout","routing-holdout-latency","Model time (API, CLI-reported)","Claude Haiku 4.5 · Claude Code","routing-holdout-latency/Model time (API, CLI-reported)","Time per routing decision, by route (Model time (API, CLI-reported))",7.522,"seconds","7.52 s",56,"\u0001","p50-p95","Claude Haiku 4.5",[7.522,23.913],"\u0001","\u0001","\u0001"],["routing-holdout","routing-holdout-latency","Model time (API, CLI-reported)","Claude Sonnet 5.5 (low) · Claude Code","routing-holdout-latency/Model time (API, CLI-reported)","Time per routing decision, by route (Model time (API, CLI-reported))",1.485,"seconds","1.49 s",56,"\u0001","p50-p95","Claude Sonnet 5.5 · effort low",[1.485,2.377],"\u0001","\u0001","\u0001"],["routing-holdout","routing-holdout-cost-per-1000","Cost","Claude Haiku 4.5 · Claude Code","routing-holdout-cost-per-1000","Cost per 1,000 unseen routing decisions",7.129,"usd","$7.13",56,"\u0001","\u0001","Claude Haiku 4.5","\u0001",true,"\u0001","\u0001"],["routing-holdout","routing-holdout-cost-per-1000","Cost","Claude Sonnet 5.5 (low) · Claude Code","routing-holdout-cost-per-1000","Cost per 1,000 unseen routing decisions",7.244,"usd","$7.24",56,"\u0001","\u0001","Claude Sonnet 5.5 · effort low","\u0001",true,"\u0001","\u0001"],["thinking-token-bill","thinking-bill-share","Median call","Claude Haiku 4.5 · Claude Code","thinking-bill-share","Reasoning share of output tokens per call on hard tasks (calculation)",91.68,"percent","91.7%",24,"\u0001","minmax","Claude Haiku 4.5",[76.46,99.27],true,"\u0001","none"],["thinking-token-bill","thinking-bill-share","Median call","Claude Fable 5.1 · Claude Code","thinking-bill-share","Reasoning share of output tokens per call on hard tasks (calculation)",64.24,"percent","64.2%",24,"\u0001","minmax","Claude Fable 5.1",[23.44,97.19],true,"\u0001","none"],["thinking-token-bill","thinking-bill-share","Median call","Claude Opus 5.5 · Claude Code","thinking-bill-share","Reasoning share of output tokens per call on hard tasks (calculation)",54.79,"percent","54.8%",24,"\u0001","minmax","Claude Opus 5.5",[29.92,95.6],true,"\u0001","none"],["thinking-token-bill","thinking-bill-share","Median call","Claude Sonnet 5.5 · Claude Code","thinking-bill-share","Reasoning share of output tokens per call on hard tasks (calculation)",54.54,"percent","54.5%",24,"\u0001","minmax","Claude Sonnet 5.5",[0,95.91],true,"\u0001","none"],["thinking-token-bill","thinking-bill-share","Median call","Claude Opus 5.5 (high) · Claude Code","thinking-bill-share","Reasoning share of output tokens per call on hard tasks (calculation)",54.43,"percent","54.4%",24,"\u0001","minmax","Claude Opus 5.5 · effort high",[36.14,96.23],true,"\u0001","none"],["thinking-token-bill","thinking-bill-cost-per-call","Reasoning (output tokens)","Claude Fable 5.1 · Claude Code","thinking-bill-cost-per-call/Reasoning (output tokens)","List-price cost per call: reasoning, remaining output and input (calculation) (Reasoning (output tokens))",0.053696,"usd","$0.054",24,"\u0001","\u0001","Claude Fable 5.1","\u0001",true,"\u0001","none"],["thinking-token-bill","thinking-bill-cost-per-call","Reasoning (output tokens)","Claude Opus 5.5 (high) · Claude Code","thinking-bill-cost-per-call/Reasoning (output tokens)","List-price cost per call: reasoning, remaining output and input (calculation) (Reasoning (output tokens))",0.017969,"usd","$0.018",24,"\u0001","\u0001","Claude Opus 5.5 · effort high","\u0001",true,"\u0001","none"],["thinking-token-bill","thinking-bill-cost-per-call","Reasoning (output tokens)","Claude Haiku 4.5 · Claude Code","thinking-bill-cost-per-call/Reasoning (output tokens)","List-price cost per call: reasoning, remaining output and input (calculation) (Reasoning (output tokens))",0.024492,"usd","$0.024",24,"\u0001","\u0001","Claude Haiku 4.5","\u0001",true,"\u0001","none"],["thinking-token-bill","thinking-bill-cost-per-call","Reasoning (output tokens)","Claude Opus 5.5 · Claude Code","thinking-bill-cost-per-call/Reasoning (output tokens)","List-price cost per call: reasoning, remaining output and input (calculation) (Reasoning (output tokens))",0.012528,"usd","$0.013",24,"\u0001","\u0001","Claude Opus 5.5","\u0001",true,"\u0001","none"],["thinking-token-bill","thinking-bill-cost-per-call","Reasoning (output tokens)","Claude Sonnet 5.5 · Claude Code","thinking-bill-cost-per-call/Reasoning (output tokens)","List-price cost per call: reasoning, remaining output and input (calculation) (Reasoning (output tokens))",0.006665,"usd","$0.0067",24,"\u0001","\u0001","Claude Sonnet 5.5","\u0001",true,"\u0001","none"],["thinking-token-bill","thinking-bill-cost-per-call","Remaining output (visible-answer estimate)","Claude Fable 5.1 · Claude Code","thinking-bill-cost-per-call/Remaining output (visible-answer estimate)","List-price cost per call: reasoning, remaining output and input (calculation) (Remaining output (visible-answer estimate))",0.018702,"usd","$0.019",24,"\u0001","\u0001","Claude Fable 5.1","\u0001",true,"\u0001","none"],["thinking-token-bill","thinking-bill-cost-per-call","Remaining output (visible-answer estimate)","Claude Opus 5.5 (high) · Claude Code","thinking-bill-cost-per-call/Remaining output (visible-answer estimate)","List-price cost per call: reasoning, remaining output and input (calculation) (Remaining output (visible-answer estimate))",0.00799,"usd","$0.0080",24,"\u0001","\u0001","Claude Opus 5.5 · effort high","\u0001",true,"\u0001","none"],["thinking-token-bill","thinking-bill-cost-per-call","Remaining output (visible-answer estimate)","Claude Haiku 4.5 · Claude Code","thinking-bill-cost-per-call/Remaining output (visible-answer estimate)","List-price cost per call: reasoning, remaining output and input (calculation) (Remaining output (visible-answer estimate))",0.001795,"usd","$0.0018",24,"\u0001","\u0001","Claude Haiku 4.5","\u0001",true,"\u0001","none"],["thinking-token-bill","thinking-bill-cost-per-call","Remaining output (visible-answer estimate)","Claude Opus 5.5 · Claude Code","thinking-bill-cost-per-call/Remaining output (visible-answer estimate)","List-price cost per call: reasoning, remaining output and input (calculation) (Remaining output (visible-answer estimate))",0.008003,"usd","$0.0080",24,"\u0001","\u0001","Claude Opus 5.5","\u0001",true,"\u0001","none"],["thinking-token-bill","thinking-bill-cost-per-call","Remaining output (visible-answer estimate)","Claude Sonnet 5.5 · Claude Code","thinking-bill-cost-per-call/Remaining output (visible-answer estimate)","List-price cost per call: reasoning, remaining output and input (calculation) (Remaining output (visible-answer estimate))",0.003672,"usd","$0.0037",24,"\u0001","\u0001","Claude Sonnet 5.5","\u0001",true,"\u0001","none"],["thinking-token-bill","thinking-bill-cost-per-call","Input (prompt, cache priced)","Claude Fable 5.1 · Claude Code","thinking-bill-cost-per-call/Input (prompt, cache priced)","List-price cost per call: reasoning, remaining output and input (calculation) (Input (prompt, cache priced))",0.02091,"usd","$0.021",24,"\u0001","\u0001","Claude Fable 5.1","\u0001",true,"\u0001","none"],["thinking-token-bill","thinking-bill-cost-per-call","Input (prompt, cache priced)","Claude Opus 5.5 (high) · Claude Code","thinking-bill-cost-per-call/Input (prompt, cache priced)","List-price cost per call: reasoning, remaining output and input (calculation) (Input (prompt, cache priced))",0.007407,"usd","$0.0074",24,"\u0001","\u0001","Claude Opus 5.5 · effort high","\u0001",true,"\u0001","none"],["thinking-token-bill","thinking-bill-cost-per-call","Input (prompt, cache priced)","Claude Haiku 4.5 · Claude Code","thinking-bill-cost-per-call/Input (prompt, cache priced)","List-price cost per call: reasoning, remaining output and input (calculation) (Input (prompt, cache priced))",0.00451,"usd","$0.0045",24,"\u0001","\u0001","Claude Haiku 4.5","\u0001",true,"\u0001","none"],["thinking-token-bill","thinking-bill-cost-per-call","Input (prompt, cache priced)","Claude Opus 5.5 · Claude Code","thinking-bill-cost-per-call/Input (prompt, cache priced)","List-price cost per call: reasoning, remaining output and input (calculation) (Input (prompt, cache priced))",0.00771,"usd","$0.0077",24,"\u0001","\u0001","Claude Opus 5.5","\u0001",true,"\u0001","none"],["thinking-token-bill","thinking-bill-cost-per-call","Input (prompt, cache priced)","Claude Sonnet 5.5 · Claude Code","thinking-bill-cost-per-call/Input (prompt, cache priced)","List-price cost per call: reasoning, remaining output and input (calculation) (Input (prompt, cache priced))",0.004012,"usd","$0.0040",24,"\u0001","\u0001","Claude Sonnet 5.5","\u0001",true,"\u0001","none"],["thinking-token-bill","thinking-bill-by-effort","Reasoning cost per strict pass","Claude Sonnet 5.5 (low) · Claude Code","thinking-bill-by-effort/Reasoning cost per strict pass","Reasoning cost per strict pass by effort, with the total (calculation) (Reasoning cost per strict pass)",0.004317,"usd","$0.0043",16,"\u0001","\u0001","Claude Sonnet 5.5 · effort low","\u0001",true,"\u0001","none"],["thinking-token-bill","thinking-bill-by-effort","Reasoning cost per strict pass","Claude Sonnet 5.5 (medium) · Claude Code","thinking-bill-by-effort/Reasoning cost per strict pass","Reasoning cost per strict pass by effort, with the total (calculation) (Reasoning cost per strict pass)",0.005946,"usd","$0.0059",16,"\u0001","\u0001","Claude Sonnet 5.5 · effort medium","\u0001",true,"\u0001","none"],["thinking-token-bill","thinking-bill-by-effort","Reasoning cost per strict pass","Claude Sonnet 5.5 (high) · Claude Code","thinking-bill-by-effort/Reasoning cost per strict pass","Reasoning cost per strict pass by effort, with the total (calculation) (Reasoning cost per strict pass)",0.009369,"usd","$0.0094",16,"\u0001","\u0001","Claude Sonnet 5.5 · effort high","\u0001",true,"\u0001","none"],["thinking-token-bill","thinking-bill-by-effort","Reasoning cost per strict pass","Claude Sonnet 5.5 · Claude Code","thinking-bill-by-effort/Reasoning cost per strict pass","Reasoning cost per strict pass by effort, with the total (calculation) (Reasoning cost per strict pass)",0.006299,"usd","$0.0063",16,"\u0001","\u0001","Claude Sonnet 5.5","\u0001",true,"\u0001","none"],["thinking-token-bill","thinking-bill-by-effort","Reasoning cost per strict pass","Claude Opus 5.5 (low) · Claude Code","thinking-bill-by-effort/Reasoning cost per strict pass","Reasoning cost per strict pass by effort, with the total (calculation) (Reasoning cost per strict pass)",0.005031,"usd","$0.0050",16,"\u0001","\u0001","Claude Opus 5.5 · effort low","\u0001",true,"\u0001","none"],["thinking-token-bill","thinking-bill-by-effort","Reasoning cost per strict pass","Claude Opus 5.5 (medium) · Claude Code","thinking-bill-by-effort/Reasoning cost per strict pass","Reasoning cost per strict pass by effort, with the total (calculation) (Reasoning cost per strict pass)",0.01344,"usd","$0.013",16,"\u0001","\u0001","Claude Opus 5.5 · effort medium","\u0001",true,"\u0001","none"],["thinking-token-bill","thinking-bill-by-effort","Reasoning cost per strict pass","Claude Opus 5.5 (high) · Claude Code","thinking-bill-by-effort/Reasoning cost per strict pass","Reasoning cost per strict pass by effort, with the total (calculation) (Reasoning cost per strict pass)",0.018034,"usd","$0.018",16,"\u0001","\u0001","Claude Opus 5.5 · effort high","\u0001",true,"\u0001","none"],["thinking-token-bill","thinking-bill-by-effort","Reasoning cost per strict pass","Claude Opus 5.5 · Claude Code","thinking-bill-by-effort/Reasoning cost per strict pass","Reasoning cost per strict pass by effort, with the total (calculation) (Reasoning cost per strict pass)",0.013104,"usd","$0.013",16,"\u0001","\u0001","Claude Opus 5.5","\u0001",true,"\u0001","none"],["thinking-token-bill","thinking-bill-by-effort","Total cost per strict pass","Claude Sonnet 5.5 (low) · Claude Code","thinking-bill-by-effort/Total cost per strict pass","Reasoning cost per strict pass by effort, with the total (calculation) (Total cost per strict pass)",0.012191,"usd","$0.012",16,"\u0001","\u0001","Claude Sonnet 5.5 · effort low","\u0001",true,"\u0001","none"],["thinking-token-bill","thinking-bill-by-effort","Total cost per strict pass","Claude Sonnet 5.5 (medium) · Claude Code","thinking-bill-by-effort/Total cost per strict pass","Reasoning cost per strict pass by effort, with the total (calculation) (Total cost per strict pass)",0.01352,"usd","$0.014",16,"\u0001","\u0001","Claude Sonnet 5.5 · effort medium","\u0001",true,"\u0001","none"],["thinking-token-bill","thinking-bill-by-effort","Total cost per strict pass","Claude Sonnet 5.5 (high) · Claude Code","thinking-bill-by-effort/Total cost per strict pass","Reasoning cost per strict pass by effort, with the total (calculation) (Total cost per strict pass)",0.016705,"usd","$0.017",16,"\u0001","\u0001","Claude Sonnet 5.5 · effort high","\u0001",true,"\u0001","none"],["thinking-token-bill","thinking-bill-by-effort","Total cost per strict pass","Claude Sonnet 5.5 · Claude Code","thinking-bill-by-effort/Total cost per strict pass","Reasoning cost per strict pass by effort, with the total (calculation) (Total cost per strict pass)",0.013978,"usd","$0.014",16,"\u0001","\u0001","Claude Sonnet 5.5","\u0001",true,"\u0001","none"],["thinking-token-bill","thinking-bill-by-effort","Total cost per strict pass","Claude Opus 5.5 (low) · Claude Code","thinking-bill-by-effort/Total cost per strict pass","Reasoning cost per strict pass by effort, with the total (calculation) (Total cost per strict pass)",0.021152,"usd","$0.021",16,"\u0001","\u0001","Claude Opus 5.5 · effort low","\u0001",true,"\u0001","none"],["thinking-token-bill","thinking-bill-by-effort","Total cost per strict pass","Claude Opus 5.5 (medium) · Claude Code","thinking-bill-by-effort/Total cost per strict pass","Reasoning cost per strict pass by effort, with the total (calculation) (Total cost per strict pass)",0.029475,"usd","$0.029",16,"\u0001","\u0001","Claude Opus 5.5 · effort medium","\u0001",true,"\u0001","none"],["thinking-token-bill","thinking-bill-by-effort","Total cost per strict pass","Claude Opus 5.5 (high) · Claude Code","thinking-bill-by-effort/Total cost per strict pass","Reasoning cost per strict pass by effort, with the total (calculation) (Total cost per strict pass)",0.033677,"usd","$0.034",16,"\u0001","\u0001","Claude Opus 5.5 · effort high","\u0001",true,"\u0001","none"],["thinking-token-bill","thinking-bill-by-effort","Total cost per strict pass","Claude Opus 5.5 · Claude Code","thinking-bill-by-effort/Total cost per strict pass","Reasoning cost per strict pass by effort, with the total (calculation) (Total cost per strict pass)",0.028925,"usd","$0.029",16,"\u0001","\u0001","Claude Opus 5.5","\u0001",true,"\u0001","none"],["thinking-token-bill","thinking-bill-short-vs-hard","Eight hard tasks","Claude Haiku 4.5 · Claude Code","thinking-bill-short-vs-hard/Eight hard tasks","Reasoning share on short tasks vs hard tasks (calculation) (Eight hard tasks)",91.68,"percent","91.7%",24,"\u0001","minmax","Claude Haiku 4.5",[76.46,99.27],true,"\u0001","none"],["thinking-token-bill","thinking-bill-short-vs-hard","Eight hard tasks","Claude Sonnet 5.5 · Claude Code","thinking-bill-short-vs-hard/Eight hard tasks","Reasoning share on short tasks vs hard tasks (calculation) (Eight hard tasks)",54.54,"percent","54.5%",24,"\u0001","minmax","Claude Sonnet 5.5",[0,95.91],true,"\u0001","none"],["thinking-token-bill","thinking-bill-short-vs-hard","Eight hard tasks","Claude Opus 5.5 · Claude Code","thinking-bill-short-vs-hard/Eight hard tasks","Reasoning share on short tasks vs hard tasks (calculation) (Eight hard tasks)",54.79,"percent","54.8%",24,"\u0001","minmax","Claude Opus 5.5",[29.92,95.6],true,"\u0001","none"],["thinking-token-bill","thinking-bill-short-vs-hard","Eight hard tasks","Claude Opus 5.5 (high) · Claude Code","thinking-bill-short-vs-hard/Eight hard tasks","Reasoning share on short tasks vs hard tasks (calculation) (Eight hard tasks)",54.43,"percent","54.4%",24,"\u0001","minmax","Claude Opus 5.5 · effort high",[36.14,96.23],true,"\u0001","none"],["thinking-token-bill","thinking-bill-short-vs-hard","Eight hard tasks","Claude Fable 5.1 · Claude Code","thinking-bill-short-vs-hard/Eight hard tasks","Reasoning share on short tasks vs hard tasks (calculation) (Eight hard tasks)",64.24,"percent","64.2%",24,"\u0001","minmax","Claude Fable 5.1",[23.44,97.19],true,"\u0001","none"],["thinking-token-bill","thinking-bill-short-vs-hard","Five short tasks","Claude Haiku 4.5 · Claude Code","thinking-bill-short-vs-hard/Five short tasks","Reasoning share on short tasks vs hard tasks (calculation) (Five short tasks)",90.19,"percent","90.2%",15,"\u0001","minmax","Claude Haiku 4.5",[73.1,97.59],true,"\u0001","none"],["thinking-token-bill","thinking-bill-short-vs-hard","Five short tasks","Claude Sonnet 5.5 · Claude Code","thinking-bill-short-vs-hard/Five short tasks","Reasoning share on short tasks vs hard tasks (calculation) (Five short tasks)",0,"percent","0%",15,"\u0001","minmax","Claude Sonnet 5.5",[0,72.75],true,"\u0001","none"],["thinking-token-bill","thinking-bill-short-vs-hard","Five short tasks","Claude Opus 5.5 · Claude Code","thinking-bill-short-vs-hard/Five short tasks","Reasoning share on short tasks vs hard tasks (calculation) (Five short tasks)",0,"percent","0%",15,"\u0001","minmax","Claude Opus 5.5",[0,93.33],true,"\u0001","none"],["thinking-token-bill","thinking-bill-short-vs-hard","Five short tasks","Claude Opus 5.5 (high) · Claude Code","thinking-bill-short-vs-hard/Five short tasks","Reasoning share on short tasks vs hard tasks (calculation) (Five short tasks)",43.59,"percent","43.6%",15,"\u0001","minmax","Claude Opus 5.5 · effort high",[0,93.33],true,"\u0001","none"],["thinking-token-bill","thinking-bill-short-vs-hard","Five short tasks","Claude Fable 5.1 · Claude Code","thinking-bill-short-vs-hard/Five short tasks","Reasoning share on short tasks vs hard tasks (calculation) (Five short tasks)",0,"percent","0%",15,"\u0001","minmax","Claude Fable 5.1",[0,74.01],true,"\u0001","none"],["llm-speed-anatomy","speed-anatomy-first-text","Time to first text","Claude Haiku 4.5 · Claude Code","speed-anatomy-first-text","Time to first text: a 250-line answer, six models",4,"seconds","4.00 s",4,"\u0001","minmax","Claude Haiku 4.5",[2.84,6.38],"\u0001","\u0001","\u0001"],["llm-speed-anatomy","speed-anatomy-first-text","Time to first text","Claude Sonnet 5.5 · Claude Code","speed-anatomy-first-text","Time to first text: a 250-line answer, six models",1.96,"seconds","1.96 s",4,"\u0001","minmax","Claude Sonnet 5.5",[0.88,4.09],"\u0001","\u0001","\u0001"],["llm-speed-anatomy","speed-anatomy-first-text","Time to first text","Claude Opus 5.5 · Claude Code","speed-anatomy-first-text","Time to first text: a 250-line answer, six models",1.97,"seconds","1.97 s",4,"\u0001","minmax","Claude Opus 5.5",[1.7,2.35],"\u0001","\u0001","\u0001"],["llm-speed-anatomy","speed-anatomy-first-text","Time to first text","Claude Fable 5.1 · Claude Code","speed-anatomy-first-text","Time to first text: a 250-line answer, six models",4.43,"seconds","4.43 s",4,"\u0001","minmax","Claude Fable 5.1",[2.27,4.64],"\u0001","\u0001","\u0001"],["llm-speed-anatomy","speed-anatomy-output-speed","Visible tokens per second","Claude Haiku 4.5 · Claude Code","speed-anatomy-output-speed","Output speed after the first text: visible tokens per second (calculation)",153.2,"tokens","153",4,"\u0001","minmax","Claude Haiku 4.5",[152.6,216.1],true,"\u0001","\u0001"],["llm-speed-anatomy","speed-anatomy-output-speed","Visible tokens per second","Claude Sonnet 5.5 · Claude Code","speed-anatomy-output-speed","Output speed after the first text: visible tokens per second (calculation)",231.7,"tokens","232",4,"\u0001","minmax","Claude Sonnet 5.5",[230.3,233],true,"\u0001","\u0001"],["llm-speed-anatomy","speed-anatomy-output-speed","Visible tokens per second","Claude Opus 5.5 · Claude Code","speed-anatomy-output-speed","Output speed after the first text: visible tokens per second (calculation)",155.5,"tokens","156",4,"\u0001","minmax","Claude Opus 5.5",[154.6,156.4],true,"\u0001","\u0001"],["llm-speed-anatomy","speed-anatomy-output-speed","Visible tokens per second","Claude Fable 5.1 · Claude Code","speed-anatomy-output-speed","Output speed after the first text: visible tokens per second (calculation)",122.6,"tokens","123",4,"\u0001","minmax","Claude Fable 5.1",[120.9,131.4],true,"\u0001","\u0001"],["llm-speed-anatomy","speed-anatomy-chars-per-second","Characters per second","Claude Haiku 4.5 · Claude Code","speed-anatomy-chars-per-second","Output speed in characters per second after the first text (calculation)",547,"count","547",3,"\u0001","minmax","Claude Haiku 4.5",[546,548],true,"\u0001","\u0001"],["llm-speed-anatomy","speed-anatomy-chars-per-second","Characters per second","Claude Sonnet 5.5 · Claude Code","speed-anatomy-chars-per-second","Output speed in characters per second after the first text (calculation)",517,"count","517",4,"\u0001","minmax","Claude Sonnet 5.5",[513,519],true,"\u0001","\u0001"],["llm-speed-anatomy","speed-anatomy-chars-per-second","Characters per second","Claude Opus 5.5 · Claude Code","speed-anatomy-chars-per-second","Output speed in characters per second after the first text (calculation)",347,"count","347",4,"\u0001","minmax","Claude Opus 5.5",[345,349],true,"\u0001","\u0001"],["llm-speed-anatomy","speed-anatomy-chars-per-second","Characters per second","Claude Fable 5.1 · Claude Code","speed-anatomy-chars-per-second","Output speed in characters per second after the first text (calculation)",273,"count","273",4,"\u0001","minmax","Claude Fable 5.1",[270,293],true,"\u0001","\u0001"],["llm-speed-anatomy","speed-anatomy-prompt-size","Claude Haiku 4.5 · Claude Code","1k","speed-anatomy-prompt-size/1k","Time to first text as the prompt grows: 1k",1.93,"seconds","1.93 s",3,"\u0001","minmax","Claude Haiku 4.5",[1.85,2.04],true,"\u0001","\u0001"],["llm-speed-anatomy","speed-anatomy-prompt-size","Claude Haiku 4.5 · Claude Code","16k","speed-anatomy-prompt-size/16k","Time to first text as the prompt grows: 16k",2.27,"seconds","2.27 s",3,"\u0001","minmax","Claude Haiku 4.5",[2.22,2.47],true,"\u0001","\u0001"],["llm-speed-anatomy","speed-anatomy-prompt-size","Claude Haiku 4.5 · Claude Code","64k","speed-anatomy-prompt-size/64k","Time to first text as the prompt grows: 64k",2.78,"seconds","2.78 s",3,"\u0001","minmax","Claude Haiku 4.5",[2.45,2.89],true,"\u0001","\u0001"],["llm-speed-anatomy","speed-anatomy-prompt-size","Claude Sonnet 5.5 · Claude Code","1k","speed-anatomy-prompt-size/1k","Time to first text as the prompt grows: 1k",1.45,"seconds","1.45 s",3,"\u0001","minmax","Claude Sonnet 5.5",[1.23,1.72],true,"\u0001","\u0001"],["llm-speed-anatomy","speed-anatomy-prompt-size","Claude Sonnet 5.5 · Claude Code","16k","speed-anatomy-prompt-size/16k","Time to first text as the prompt grows: 16k",1.78,"seconds","1.78 s",3,"\u0001","minmax","Claude Sonnet 5.5",[1.64,2.11],true,"\u0001","\u0001"],["llm-speed-anatomy","speed-anatomy-prompt-size","Claude Sonnet 5.5 · Claude Code","64k","speed-anatomy-prompt-size/64k","Time to first text as the prompt grows: 64k",3.07,"seconds","3.07 s",3,"\u0001","minmax","Claude Sonnet 5.5",[1.38,3.61],true,"\u0001","\u0001"],["llm-speed-anatomy","speed-anatomy-prompt-size","Claude Opus 5.5 · Claude Code","1k","speed-anatomy-prompt-size/1k","Time to first text as the prompt grows: 1k",1.51,"seconds","1.51 s",3,"\u0001","minmax","Claude Opus 5.5",[1.46,2.01],true,"\u0001","\u0001"],["llm-speed-anatomy","speed-anatomy-prompt-size","Claude Opus 5.5 · Claude Code","16k","speed-anatomy-prompt-size/16k","Time to first text as the prompt grows: 16k",1.74,"seconds","1.74 s",3,"\u0001","minmax","Claude Opus 5.5",[1.7,2.97],true,"\u0001","\u0001"],["llm-speed-anatomy","speed-anatomy-prompt-size","Claude Opus 5.5 · Claude Code","64k","speed-anatomy-prompt-size/64k","Time to first text as the prompt grows: 64k",1.79,"seconds","1.79 s",3,"\u0001","minmax","Claude Opus 5.5",[1.72,3.72],true,"\u0001","\u0001"],["llm-speed-anatomy","speed-anatomy-total-by-size","1k prompt","Claude Haiku 4.5 · Claude Code","speed-anatomy-total-by-size/1k prompt","Total time per call by prompt size (1k prompt)",2.34,"seconds","2.34 s",3,"\u0001","minmax","Claude Haiku 4.5",[2.22,2.46],"\u0001","\u0001","\u0001"],["llm-speed-anatomy","speed-anatomy-total-by-size","1k prompt","Claude Sonnet 5.5 · Claude Code","speed-anatomy-total-by-size/1k prompt","Total time per call by prompt size (1k prompt)",1.78,"seconds","1.78 s",3,"\u0001","minmax","Claude Sonnet 5.5",[1.57,2.12],"\u0001","\u0001","\u0001"],["llm-speed-anatomy","speed-anatomy-total-by-size","1k prompt","Claude Opus 5.5 · Claude Code","speed-anatomy-total-by-size/1k prompt","Total time per call by prompt size (1k prompt)",1.83,"seconds","1.83 s",3,"\u0001","minmax","Claude Opus 5.5",[1.82,2.41],"\u0001","\u0001","\u0001"],["llm-speed-anatomy","speed-anatomy-total-by-size","16k prompt","Claude Haiku 4.5 · Claude Code","speed-anatomy-total-by-size/16k prompt","Total time per call by prompt size (16k prompt)",2.79,"seconds","2.79 s",3,"\u0001","minmax","Claude Haiku 4.5",[2.58,2.84],"\u0001","\u0001","\u0001"],["llm-speed-anatomy","speed-anatomy-total-by-size","16k prompt","Claude Sonnet 5.5 · Claude Code","speed-anatomy-total-by-size/16k prompt","Total time per call by prompt size (16k prompt)",2.1,"seconds","2.10 s",3,"\u0001","minmax","Claude Sonnet 5.5",[1.98,2.48],"\u0001","\u0001","\u0001"],["llm-speed-anatomy","speed-anatomy-total-by-size","16k prompt","Claude Opus 5.5 · Claude Code","speed-anatomy-total-by-size/16k prompt","Total time per call by prompt size (16k prompt)",2.36,"seconds","2.36 s",3,"\u0001","minmax","Claude Opus 5.5",[2.11,3.4],"\u0001","\u0001","\u0001"],["llm-speed-anatomy","speed-anatomy-total-by-size","64k prompt","Claude Haiku 4.5 · Claude Code","speed-anatomy-total-by-size/64k prompt","Total time per call by prompt size (64k prompt)",3.13,"seconds","3.13 s",3,"\u0001","minmax","Claude Haiku 4.5",[2.84,3.28],"\u0001","\u0001","\u0001"],["llm-speed-anatomy","speed-anatomy-total-by-size","64k prompt","Claude Sonnet 5.5 · Claude Code","speed-anatomy-total-by-size/64k prompt","Total time per call by prompt size (64k prompt)",3.44,"seconds","3.44 s",3,"\u0001","minmax","Claude Sonnet 5.5",[1.74,4.38],"\u0001","\u0001","\u0001"],["llm-speed-anatomy","speed-anatomy-total-by-size","64k prompt","Claude Opus 5.5 · Claude Code","speed-anatomy-total-by-size/64k prompt","Total time per call by prompt size (64k prompt)",2.35,"seconds","2.35 s",3,"\u0001","minmax","Claude Opus 5.5",[2.26,4.29],"\u0001","\u0001","\u0001"],["llm-speed-anatomy","speed-anatomy-lookup-correct","Exact answer","Claude Haiku 4.5 · Claude Code","speed-anatomy-lookup-correct","Exact lookup answers at the 1k, 16k and 64k prompt-size targets",1,"rate","100% (9/9)",9,[0.7009,1],"ci95","Claude Haiku 4.5","\u0001","\u0001","\u0001","higher"],["llm-speed-anatomy","speed-anatomy-lookup-correct","Exact answer","Claude Sonnet 5.5 · Claude Code","speed-anatomy-lookup-correct","Exact lookup answers at the 1k, 16k and 64k prompt-size targets",1,"rate","100% (9/9)",9,[0.7009,1],"ci95","Claude Sonnet 5.5","\u0001","\u0001","\u0001","higher"],["llm-speed-anatomy","speed-anatomy-lookup-correct","Exact answer","Claude Opus 5.5 · Claude Code","speed-anatomy-lookup-correct","Exact lookup answers at the 1k, 16k and 64k prompt-size targets",0.5556,"rate","56% (5/9)",9,[0.2667,0.8112],"ci95","Claude Opus 5.5","\u0001","\u0001","\u0001","higher"],["haiku-retry-or-escalate","retry-escalate-call-cost-by-task","Claude Haiku 4.5 · Claude Code","Interval merge fix","retry-escalate-call-cost-by-task/Interval merge fix","List-price cost of one call, Haiku 4.5 vs Sonnet 5.5, on each hard task (calculation): Interval merge fix",0.01789,"usd","$0.018",3,"\u0001","\u0001","Claude Haiku 4.5","\u0001",true,"\u0001","\u0001"],["haiku-retry-or-escalate","retry-escalate-call-cost-by-task","Claude Haiku 4.5 · Claude Code","DST day length","retry-escalate-call-cost-by-task/DST day length","List-price cost of one call, Haiku 4.5 vs Sonnet 5.5, on each hard task (calculation): DST day length",0.0293,"usd","$0.029",3,"\u0001","\u0001","Claude Haiku 4.5","\u0001",true,"\u0001","\u0001"],["haiku-retry-or-escalate","retry-escalate-call-cost-by-task","Claude Haiku 4.5 · Claude Code","CSV parser","retry-escalate-call-cost-by-task/CSV parser","List-price cost of one call, Haiku 4.5 vs Sonnet 5.5, on each hard task (calculation): CSV parser",0.02919,"usd","$0.029",3,"\u0001","\u0001","Claude Haiku 4.5","\u0001",true,"\u0001","\u0001"],["haiku-retry-or-escalate","retry-escalate-call-cost-by-task","Claude Haiku 4.5 · Claude Code","Event-loop order","retry-escalate-call-cost-by-task/Event-loop order","List-price cost of one call, Haiku 4.5 vs Sonnet 5.5, on each hard task (calculation): Event-loop order",0.03708,"usd","$0.037",3,"\u0001","\u0001","Claude Haiku 4.5","\u0001",true,"\u0001","\u0001"],["haiku-retry-or-escalate","retry-escalate-call-cost-by-task","Claude Haiku 4.5 · Claude Code","Room schedule","retry-escalate-call-cost-by-task/Room schedule","List-price cost of one call, Haiku 4.5 vs Sonnet 5.5, on each hard task (calculation): Room schedule",0.03578,"usd","$0.036",3,"\u0001","\u0001","Claude Haiku 4.5","\u0001",true,"\u0001","\u0001"],["haiku-retry-or-escalate","retry-escalate-call-cost-by-task","Claude Haiku 4.5 · Claude Code","SemVer regex","retry-escalate-call-cost-by-task/SemVer regex","List-price cost of one call, Haiku 4.5 vs Sonnet 5.5, on each hard task (calculation): SemVer regex",0.04217,"usd","$0.042",3,"\u0001","\u0001","Claude Haiku 4.5","\u0001",true,"\u0001","\u0001"],["haiku-retry-or-escalate","retry-escalate-call-cost-by-task","Claude Haiku 4.5 · Claude Code","Money refactor","retry-escalate-call-cost-by-task/Money refactor","List-price cost of one call, Haiku 4.5 vs Sonnet 5.5, on each hard task (calculation): Money refactor",0.02039,"usd","$0.020",3,"\u0001","\u0001","Claude Haiku 4.5","\u0001",true,"\u0001","\u0001"],["haiku-retry-or-escalate","retry-escalate-call-cost-by-task","Claude Haiku 4.5 · Claude Code","SQLite report query","retry-escalate-call-cost-by-task/SQLite report query","List-price cost of one call, Haiku 4.5 vs Sonnet 5.5, on each hard task (calculation): SQLite report query",0.03648,"usd","$0.036",3,"\u0001","\u0001","Claude Haiku 4.5","\u0001",true,"\u0001","\u0001"],["haiku-retry-or-escalate","retry-escalate-call-cost-by-task","Claude Sonnet 5.5 · Claude Code","Interval merge fix","retry-escalate-call-cost-by-task/Interval merge fix","List-price cost of one call, Haiku 4.5 vs Sonnet 5.5, on each hard task (calculation): Interval merge fix",0.00557,"usd","$0.0056",3,"\u0001","\u0001","Claude Sonnet 5.5","\u0001",true,"\u0001","\u0001"],["haiku-retry-or-escalate","retry-escalate-call-cost-by-task","Claude Sonnet 5.5 · Claude Code","DST day length","retry-escalate-call-cost-by-task/DST day length","List-price cost of one call, Haiku 4.5 vs Sonnet 5.5, on each hard task (calculation): DST day length",0.02532,"usd","$0.025",3,"\u0001","\u0001","Claude Sonnet 5.5","\u0001",true,"\u0001","\u0001"],["haiku-retry-or-escalate","retry-escalate-call-cost-by-task","Claude Sonnet 5.5 · Claude Code","CSV parser","retry-escalate-call-cost-by-task/CSV parser","List-price cost of one call, Haiku 4.5 vs Sonnet 5.5, on each hard task (calculation): CSV parser",0.01464,"usd","$0.015",3,"\u0001","\u0001","Claude Sonnet 5.5","\u0001",true,"\u0001","\u0001"],["haiku-retry-or-escalate","retry-escalate-call-cost-by-task","Claude Sonnet 5.5 · Claude Code","Event-loop order","retry-escalate-call-cost-by-task/Event-loop order","List-price cost of one call, Haiku 4.5 vs Sonnet 5.5, on each hard task (calculation): Event-loop order",0.01588,"usd","$0.016",3,"\u0001","\u0001","Claude Sonnet 5.5","\u0001",true,"\u0001","\u0001"],["haiku-retry-or-escalate","retry-escalate-call-cost-by-task","Claude Sonnet 5.5 · Claude Code","Room schedule","retry-escalate-call-cost-by-task/Room schedule","List-price cost of one call, Haiku 4.5 vs Sonnet 5.5, on each hard task (calculation): Room schedule",0.01243,"usd","$0.012",3,"\u0001","\u0001","Claude Sonnet 5.5","\u0001",true,"\u0001","\u0001"],["haiku-retry-or-escalate","retry-escalate-call-cost-by-task","Claude Sonnet 5.5 · Claude Code","SemVer regex","retry-escalate-call-cost-by-task/SemVer regex","List-price cost of one call, Haiku 4.5 vs Sonnet 5.5, on each hard task (calculation): SemVer regex",0.00514,"usd","$0.0051",3,"\u0001","\u0001","Claude Sonnet 5.5","\u0001",true,"\u0001","\u0001"],["haiku-retry-or-escalate","retry-escalate-call-cost-by-task","Claude Sonnet 5.5 · Claude Code","Money refactor","retry-escalate-call-cost-by-task/Money refactor","List-price cost of one call, Haiku 4.5 vs Sonnet 5.5, on each hard task (calculation): Money refactor",0.00961,"usd","$0.0096",3,"\u0001","\u0001","Claude Sonnet 5.5","\u0001",true,"\u0001","\u0001"],["haiku-retry-or-escalate","retry-escalate-call-cost-by-task","Claude Sonnet 5.5 · Claude Code","SQLite report query","retry-escalate-call-cost-by-task/SQLite report query","List-price cost of one call, Haiku 4.5 vs Sonnet 5.5, on each hard task (calculation): SQLite report query",0.01719,"usd","$0.017",3,"\u0001","\u0001","Claude Sonnet 5.5","\u0001",true,"\u0001","\u0001"],["harder-tasks-head-to-head","harder-h2h-pass-rate","Strict pass","Claude Opus 5.5 · Claude Code","harder-h2h-pass-rate/Strict pass","Pass rate on 4 harder tasks (Strict pass)",0.4167,"rate","42% (5/12)",12,[0.1933,0.6805],"ci95","Claude Opus 5.5","\u0001","\u0001","\u0001","higher"],["harder-tasks-head-to-head","harder-h2h-pass-rate","Strict pass","Claude Sonnet 5.5 · Claude Code","harder-h2h-pass-rate/Strict pass","Pass rate on 4 harder tasks (Strict pass)",0.375,"rate","38% (6/16)",16,[0.1848,0.6136],"ci95","Claude Sonnet 5.5","\u0001","\u0001","\u0001","higher"],["harder-tasks-head-to-head","harder-h2h-pass-rate","Strict pass","Claude Haiku 4.5 · Claude Code","harder-h2h-pass-rate/Strict pass","Pass rate on 4 harder tasks (Strict pass)",0,"rate","0% (0/12)",12,[0,0.2425],"ci95","Claude Haiku 4.5","\u0001","\u0001","\u0001","higher"],["harder-tasks-head-to-head","harder-h2h-pass-rate","Lenient (format misses counted)","Claude Opus 5.5 · Claude Code","harder-h2h-pass-rate/Lenient (format misses counted)","Pass rate on 4 harder tasks (Lenient (format misses counted))",0.5,"rate","50% (6/12)",12,[0.2538,0.7462],"ci95","Claude Opus 5.5","\u0001","\u0001","\u0001","higher"],["harder-tasks-head-to-head","harder-h2h-pass-rate","Lenient (format misses counted)","Claude Sonnet 5.5 · Claude Code","harder-h2h-pass-rate/Lenient (format misses counted)","Pass rate on 4 harder tasks (Lenient (format misses counted))",0.375,"rate","38% (6/16)",16,[0.1848,0.6136],"ci95","Claude Sonnet 5.5","\u0001","\u0001","\u0001","higher"],["harder-tasks-head-to-head","harder-h2h-pass-rate","Lenient (format misses counted)","Claude Haiku 4.5 · Claude Code","harder-h2h-pass-rate/Lenient (format misses counted)","Pass rate on 4 harder tasks (Lenient (format misses counted))",0,"rate","0% (0/12)",12,[0,0.2425],"ci95","Claude Haiku 4.5","\u0001","\u0001","\u0001","higher"],["harder-tasks-head-to-head","harder-h2h-tool-attempts","Tool attempt","Claude Opus 5.5 · Claude Code","harder-h2h-tool-attempts","Calls that tried a tool although tools were off",0.4167,"rate","42% (5/12)",12,[0.1933,0.6805],"ci95","Claude Opus 5.5","\u0001","\u0001","\u0001","none"],["harder-tasks-head-to-head","harder-h2h-tool-attempts","Tool attempt","Claude Sonnet 5.5 · Claude Code","harder-h2h-tool-attempts","Calls that tried a tool although tools were off",0.3125,"rate","31% (5/16)",16,[0.1416,0.556],"ci95","Claude Sonnet 5.5","\u0001","\u0001","\u0001","none"],["harder-tasks-head-to-head","harder-h2h-tool-attempts","Tool attempt","Claude Haiku 4.5 · Claude Code","harder-h2h-tool-attempts","Calls that tried a tool although tools were off",0.0833,"rate","8% (1/12)",12,[0.0149,0.3539],"ci95","Claude Haiku 4.5","\u0001","\u0001","\u0001","none"],["harder-tasks-head-to-head","harder-h2h-pass-by-task","Claude Opus 5.5 · Claude Code","10x10 nonogram","harder-h2h-pass-by-task/10x10 nonogram","Strict pass rate by task: 10x10 nonogram",1,"rate","100% (3/3)",3,[0.4385,1],"ci95","Claude Opus 5.5","\u0001","\u0001","\u0001","higher"],["harder-tasks-head-to-head","harder-h2h-pass-by-task","Claude Opus 5.5 · Claude Code","Sudoku, 22 givens","harder-h2h-pass-by-task/Sudoku, 22 givens","Strict pass rate by task: Sudoku, 22 givens",0,"rate","0% (0/3)",3,[0,0.5615],"ci95","Claude Opus 5.5","\u0001","\u0001","\u0001","higher"],["harder-tasks-head-to-head","harder-h2h-pass-by-task","Claude Opus 5.5 · Claude Code","6x6 Skyscrapers","harder-h2h-pass-by-task/6x6 Skyscrapers","Strict pass rate by task: 6x6 Skyscrapers",0.3333,"rate","33% (1/3)",3,[0.0615,0.7923],"ci95","Claude Opus 5.5","\u0001","\u0001","\u0001","higher"],["harder-tasks-head-to-head","harder-h2h-pass-by-task","Claude Opus 5.5 · Claude Code","Seeded shuffle output","harder-h2h-pass-by-task/Seeded shuffle output","Strict pass rate by task: Seeded shuffle output",0.3333,"rate","33% (1/3)",3,[0.0615,0.7923],"ci95","Claude Opus 5.5","\u0001","\u0001","\u0001","higher"],["harder-tasks-head-to-head","harder-h2h-pass-by-task","Claude Sonnet 5.5 · Claude Code","10x10 nonogram","harder-h2h-pass-by-task/10x10 nonogram","Strict pass rate by task: 10x10 nonogram",1,"rate","100% (4/4)",4,[0.5101,1],"ci95","Claude Sonnet 5.5","\u0001","\u0001","\u0001","higher"],["harder-tasks-head-to-head","harder-h2h-pass-by-task","Claude Sonnet 5.5 · Claude Code","Sudoku, 22 givens","harder-h2h-pass-by-task/Sudoku, 22 givens","Strict pass rate by task: Sudoku, 22 givens",0,"rate","0% (0/4)",4,[0,0.4899],"ci95","Claude Sonnet 5.5","\u0001","\u0001","\u0001","higher"],["harder-tasks-head-to-head","harder-h2h-pass-by-task","Claude Sonnet 5.5 · Claude Code","6x6 Skyscrapers","harder-h2h-pass-by-task/6x6 Skyscrapers","Strict pass rate by task: 6x6 Skyscrapers",0,"rate","0% (0/4)",4,[0,0.4899],"ci95","Claude Sonnet 5.5","\u0001","\u0001","\u0001","higher"],["harder-tasks-head-to-head","harder-h2h-pass-by-task","Claude Sonnet 5.5 · Claude Code","Seeded shuffle output","harder-h2h-pass-by-task/Seeded shuffle output","Strict pass rate by task: Seeded shuffle output",0.5,"rate","50% (2/4)",4,[0.15,0.85],"ci95","Claude Sonnet 5.5","\u0001","\u0001","\u0001","higher"],["harder-tasks-head-to-head","harder-h2h-pass-by-task","Claude Haiku 4.5 · Claude Code","10x10 nonogram","harder-h2h-pass-by-task/10x10 nonogram","Strict pass rate by task: 10x10 nonogram",0,"rate","0% (0/3)",3,[0,0.5615],"ci95","Claude Haiku 4.5","\u0001","\u0001","\u0001","higher"],["harder-tasks-head-to-head","harder-h2h-pass-by-task","Claude Haiku 4.5 · Claude Code","Sudoku, 22 givens","harder-h2h-pass-by-task/Sudoku, 22 givens","Strict pass rate by task: Sudoku, 22 givens",0,"rate","0% (0/3)",3,[0,0.5615],"ci95","Claude Haiku 4.5","\u0001","\u0001","\u0001","higher"],["harder-tasks-head-to-head","harder-h2h-pass-by-task","Claude Haiku 4.5 · Claude Code","6x6 Skyscrapers","harder-h2h-pass-by-task/6x6 Skyscrapers","Strict pass rate by task: 6x6 Skyscrapers",0,"rate","0% (0/3)",3,[0,0.5615],"ci95","Claude Haiku 4.5","\u0001","\u0001","\u0001","higher"],["harder-tasks-head-to-head","harder-h2h-pass-by-task","Claude Haiku 4.5 · Claude Code","Seeded shuffle output","harder-h2h-pass-by-task/Seeded shuffle output","Strict pass rate by task: Seeded shuffle output",0,"rate","0% (0/3)",3,[0,0.5615],"ci95","Claude Haiku 4.5","\u0001","\u0001","\u0001","higher"],["harder-tasks-head-to-head","harder-h2h-total-latency","Total time per call","Claude Opus 5.5 · Claude Code","harder-h2h-total-latency","Total time per call on harder tasks",80.34,"seconds","80.3 s",9,"\u0001","minmax","Claude Opus 5.5",[3.82,279.5],"\u0001","\u0001","\u0001"],["harder-tasks-head-to-head","harder-h2h-total-latency","Total time per call","Claude Sonnet 5.5 · Claude Code","harder-h2h-total-latency","Total time per call on harder tasks",70.43,"seconds","70.4 s",12,"\u0001","minmax","Claude Sonnet 5.5",[4.32,210.08],"\u0001","\u0001","\u0001"],["harder-tasks-head-to-head","harder-h2h-total-latency","Total time per call","Claude Haiku 4.5 · Claude Code","harder-h2h-total-latency","Total time per call on harder tasks",108.98,"seconds","109.0 s",10,"\u0001","minmax","Claude Haiku 4.5",[25.73,223.95],"\u0001","\u0001","\u0001"],["harder-tasks-head-to-head","harder-h2h-output-tokens","Output tokens","Claude Opus 5.5 · Claude Code","harder-h2h-output-tokens/Output tokens","Output tokens per call on harder tasks (Output tokens)",8420,"tokens","8,420",9,"\u0001","minmax","Claude Opus 5.5",[279,40044],"\u0001","\u0001","none"],["harder-tasks-head-to-head","harder-h2h-output-tokens","Output tokens","Claude Sonnet 5.5 · Claude Code","harder-h2h-output-tokens/Output tokens","Output tokens per call on harder tasks (Output tokens)",9287,"tokens","9,287",12,"\u0001","minmax","Claude Sonnet 5.5",[407,27921],"\u0001","\u0001","none"],["harder-tasks-head-to-head","harder-h2h-output-tokens","Output tokens","Claude Haiku 4.5 · Claude Code","harder-h2h-output-tokens/Output tokens","Output tokens per call on harder tasks (Output tokens)",12508,"tokens","12,508",10,"\u0001","minmax","Claude Haiku 4.5",[2965,26532],"\u0001","\u0001","none"],["harder-tasks-head-to-head","harder-h2h-cost-per-pass","Cost per strict pass","Claude Sonnet 5.5 · Claude Code","harder-h2h-cost-per-pass","List-price cost per strict pass on harder tasks (calculation)",0.23843,"usd","$0.24",16,"\u0001","\u0001","Claude Sonnet 5.5","\u0001",true,"\u0001","\u0001"],["harder-tasks-head-to-head","harder-h2h-cost-per-pass","Cost per strict pass","Claude Opus 5.5 · Claude Code","harder-h2h-cost-per-pass","List-price cost per strict pass on harder tasks (calculation)",0.59333,"usd","$0.59",12,"\u0001","\u0001","Claude Opus 5.5","\u0001",true,"\u0001","\u0001"]]}],["codex-cli","Codex CLI","OpenAI","cli","OpenAI’s coding CLI. Each measurement pairs it with one GPT model and effort; the context names them.",["Codex CLI"],{"$k":["studySlug","chartId","series","point","metric","label","value","unit","display","n","ci","spanKind","context","range","calculation","statId","polarity"],"$r":[["model-head-to-head","h2h-pass-rate","Pass rate","GPT-6.1 Sol (high) · Codex CLI","h2h-pass-rate","Pass rate on five validated tasks",1,"rate","100% (15/15)",15,[0.7961,1],"ci95","GPT-6.1 Sol · effort high · five short validated tasks","\u0001","\u0001","\u0001","\u0001"],["model-head-to-head","h2h-pass-rate","Pass rate","GPT-6.1 Sol (medium) · Codex CLI","h2h-pass-rate","Pass rate on five validated tasks",1,"rate","100% (15/15)",15,[0.7961,1],"ci95","GPT-6.1 Sol · effort medium · five short validated tasks","\u0001","\u0001","\u0001","\u0001"],["model-head-to-head","h2h-pass-rate","Pass rate","GPT-6.1 Sol (low) · Codex CLI","h2h-pass-rate","Pass rate on five validated tasks",1,"rate","100% (10/10)",10,[0.7225,1],"ci95","GPT-6.1 Sol · effort low · five short validated tasks","\u0001","\u0001","\u0001","\u0001"],["model-head-to-head","h2h-total-latency","Total time per call","GPT-6.1 Sol (high) · Codex CLI","h2h-total-latency","Total time per call",5.6,"seconds","5.60 s",15,"\u0001","minmax","GPT-6.1 Sol · effort high · five short validated tasks",[4.05,19.52],"\u0001","\u0001","\u0001"],["model-head-to-head","h2h-total-latency","Total time per call","GPT-6.1 Sol (medium) · Codex CLI","h2h-total-latency","Total time per call",5.65,"seconds","5.65 s",15,"\u0001","minmax","GPT-6.1 Sol · effort medium · five short validated tasks",[4.1,25.46],"\u0001","\u0001","\u0001"],["model-head-to-head","h2h-total-latency","Total time per call","GPT-6.1 Sol (low) · Codex CLI","h2h-total-latency","Total time per call",6.26,"seconds","6.26 s",10,"\u0001","minmax","GPT-6.1 Sol · effort low · five short validated tasks",[4.65,10.47],"\u0001","\u0001","\u0001"],["model-head-to-head","h2h-first-useful-latency","Time to first useful output","GPT-6.1 Sol (high) · Codex CLI","h2h-first-useful-latency","Time to first useful output",5.32,"seconds","5.32 s",15,"\u0001","minmax","GPT-6.1 Sol · effort high · five short validated tasks",[3.64,16.37],"\u0001","\u0001","\u0001"],["model-head-to-head","h2h-first-useful-latency","Time to first useful output","GPT-6.1 Sol (medium) · Codex CLI","h2h-first-useful-latency","Time to first useful output",5.05,"seconds","5.05 s",15,"\u0001","minmax","GPT-6.1 Sol · effort medium · five short validated tasks",[3.36,17.82],"\u0001","\u0001","\u0001"],["model-head-to-head","h2h-first-useful-latency","Time to first useful output","GPT-6.1 Sol (low) · Codex CLI","h2h-first-useful-latency","Time to first useful output",5.14,"seconds","5.14 s",10,"\u0001","minmax","GPT-6.1 Sol · effort low · five short validated tasks",[4.02,8.5],"\u0001","\u0001","\u0001"],["model-head-to-head","h2h-input-tokens","Cache read","GPT-6.1 Sol (high) · Codex CLI","h2h-input-tokens/Cache read","Input tokens per call: what the CLI sends (Cache read)",6716,"tokens","6,716",15,"\u0001","\u0001","GPT-6.1 Sol · effort high · five short validated tasks","\u0001","\u0001","\u0001","\u0001"],["model-head-to-head","h2h-input-tokens","Cache read","GPT-6.1 Sol (medium) · Codex CLI","h2h-input-tokens/Cache read","Input tokens per call: what the CLI sends (Cache read)",5180,"tokens","5,180",15,"\u0001","\u0001","GPT-6.1 Sol · effort medium · five short validated tasks","\u0001","\u0001","\u0001","\u0001"],["model-head-to-head","h2h-input-tokens","Cache read","GPT-6.1 Sol (low) · Codex CLI","h2h-input-tokens/Cache read","Input tokens per call: what the CLI sends (Cache read)",8064,"tokens","8,064",10,"\u0001","\u0001","GPT-6.1 Sol · effort low · five short validated tasks","\u0001","\u0001","\u0001","\u0001"],["model-head-to-head","h2h-input-tokens","Other input","GPT-6.1 Sol (high) · Codex CLI","h2h-input-tokens/Other input","Input tokens per call: what the CLI sends (Other input)",5406,"tokens","5,406",15,"\u0001","\u0001","GPT-6.1 Sol · effort high · five short validated tasks","\u0001","\u0001","\u0001","\u0001"],["model-head-to-head","h2h-input-tokens","Other input","GPT-6.1 Sol (medium) · Codex CLI","h2h-input-tokens/Other input","Input tokens per call: what the CLI sends (Other input)",6943,"tokens","6,943",15,"\u0001","\u0001","GPT-6.1 Sol · effort medium · five short validated tasks","\u0001","\u0001","\u0001","\u0001"],["model-head-to-head","h2h-input-tokens","Other input","GPT-6.1 Sol (low) · Codex CLI","h2h-input-tokens/Other input","Input tokens per call: what the CLI sends (Other input)",4059,"tokens","4,059",10,"\u0001","\u0001","GPT-6.1 Sol · effort low · five short validated tasks","\u0001","\u0001","\u0001","\u0001"],["model-head-to-head","h2h-output-tokens","Output tokens","GPT-6.1 Sol (high) · Codex CLI","h2h-output-tokens/Output tokens","Output tokens per call (Output tokens)",42,"tokens","42",15,"\u0001","\u0001","GPT-6.1 Sol · effort high · five short validated tasks","\u0001","\u0001","\u0001","\u0001"],["model-head-to-head","h2h-output-tokens","Output tokens","GPT-6.1 Sol (medium) · Codex CLI","h2h-output-tokens/Output tokens","Output tokens per call (Output tokens)",42,"tokens","42",15,"\u0001","\u0001","GPT-6.1 Sol · effort medium · five short validated tasks","\u0001","\u0001","\u0001","\u0001"],["model-head-to-head","h2h-output-tokens","Output tokens","GPT-6.1 Sol (low) · Codex CLI","h2h-output-tokens/Output tokens","Output tokens per call (Output tokens)",42,"tokens","42",10,"\u0001","\u0001","GPT-6.1 Sol · effort low · five short validated tasks","\u0001","\u0001","\u0001","\u0001"],["model-head-to-head","h2h-list-price-per-call","Cost per call","GPT-6.1 Sol (high) · Codex CLI","h2h-list-price-per-call","List-price cost per call (calculation)",0.01047,"usd","$0.010",15,"\u0001","minmax","GPT-6.1 Sol · effort high · five short validated tasks",[0.0066,0.02812],true,"\u0001","\u0001"],["model-head-to-head","h2h-list-price-per-call","Cost per call","GPT-6.1 Sol (medium) · Codex CLI","h2h-list-price-per-call","List-price cost per call (calculation)",0.01018,"usd","$0.010",15,"\u0001","minmax","GPT-6.1 Sol · effort medium · five short validated tasks",[0.0054,0.02686],true,"\u0001","\u0001"],["model-head-to-head","h2h-list-price-per-call","Cost per call","GPT-6.1 Sol (low) · Codex CLI","h2h-list-price-per-call","List-price cost per call (calculation)",0.00769,"usd","$0.0077",10,"\u0001","minmax","GPT-6.1 Sol · effort low · five short validated tasks",[0.00742,0.02649],true,"\u0001","\u0001"],["model-head-to-head","h2h-cost-per-pass","Cost per pass","GPT-6.1 Sol (low) · Codex CLI","h2h-cost-per-pass","List-price cost per passing answer (calculation)",0.00998,"usd","$0.010",10,"\u0001","\u0001","GPT-6.1 Sol · effort low · five short validated tasks","\u0001",true,"\u0001","\u0001"],["model-head-to-head","h2h-cost-per-pass","Cost per pass","GPT-6.1 Sol (high) · Codex CLI","h2h-cost-per-pass","List-price cost per passing answer (calculation)",0.01322,"usd","$0.013",15,"\u0001","\u0001","GPT-6.1 Sol · effort high · five short validated tasks","\u0001",true,"\u0001","\u0001"],["model-head-to-head","h2h-cost-per-pass","Cost per pass","GPT-6.1 Sol (medium) · Codex CLI","h2h-cost-per-pass","List-price cost per passing answer (calculation)",0.01564,"usd","$0.016",15,"\u0001","\u0001","GPT-6.1 Sol · effort medium · five short validated tasks","\u0001",true,"\u0001","\u0001"],["hard-model-head-to-head","hard-h2h-pass-rate","Strict pass","GPT-6.1 Sol (medium) · Codex CLI","hard-h2h-pass-rate/Strict pass","Pass rate on eight hard tasks (Strict pass)",1,"rate","100% (16/16)",16,[0.8064,1],"ci95","GPT-6.1 Sol · effort medium · eight hard validated tasks","\u0001","\u0001","\u0001","\u0001"],["hard-model-head-to-head","hard-h2h-pass-rate","Strict pass","GPT-6.1 Sol (high) · Codex CLI","hard-h2h-pass-rate/Strict pass","Pass rate on eight hard tasks (Strict pass)",1,"rate","100% (16/16)",16,[0.8064,1],"ci95","GPT-6.1 Sol · effort high · eight hard validated tasks","\u0001","\u0001","\u0001","\u0001"],["hard-model-head-to-head","hard-h2h-pass-rate","Lenient (format misses counted)","GPT-6.1 Sol (medium) · Codex CLI","hard-h2h-pass-rate/Lenient (format misses counted)","Pass rate on eight hard tasks (Lenient (format misses counted))",1,"rate","100% (16/16)",16,[0.8064,1],"ci95","GPT-6.1 Sol · effort medium · eight hard validated tasks","\u0001","\u0001","\u0001","\u0001"],["hard-model-head-to-head","hard-h2h-pass-rate","Lenient (format misses counted)","GPT-6.1 Sol (high) · Codex CLI","hard-h2h-pass-rate/Lenient (format misses counted)","Pass rate on eight hard tasks (Lenient (format misses counted))",1,"rate","100% (16/16)",16,[0.8064,1],"ci95","GPT-6.1 Sol · effort high · eight hard validated tasks","\u0001","\u0001","\u0001","\u0001"],["hard-model-head-to-head","hard-h2h-total-latency","Total time per call on hard tasks (separate batches)","GPT-6.1 Sol (medium) · Codex CLI","hard-h2h-total-latency","Total time per call on hard tasks (separate batches)",13.11,"seconds","13.1 s",16,"\u0001","minmax","GPT-6.1 Sol · effort medium · eight hard validated tasks",[8.54,61.6],"\u0001","\u0001","\u0001"],["hard-model-head-to-head","hard-h2h-total-latency","Total time per call on hard tasks (separate batches)","GPT-6.1 Sol (high) · Codex CLI","hard-h2h-total-latency","Total time per call on hard tasks (separate batches)",18.12,"seconds","18.1 s",16,"\u0001","minmax","GPT-6.1 Sol · effort high · eight hard validated tasks",[11.67,92.21],"\u0001","\u0001","\u0001"],["hard-model-head-to-head","hard-h2h-first-useful-latency","Time to first useful output on hard tasks","GPT-6.1 Sol (medium) · Codex CLI","hard-h2h-first-useful-latency","Time to first useful output on hard tasks",10.23,"seconds","10.2 s",16,"\u0001","minmax","GPT-6.1 Sol · effort medium · eight hard validated tasks",[6.09,40.41],"\u0001","\u0001","\u0001"],["hard-model-head-to-head","hard-h2h-first-useful-latency","Time to first useful output on hard tasks","GPT-6.1 Sol (high) · Codex CLI","hard-h2h-first-useful-latency","Time to first useful output on hard tasks",12.69,"seconds","12.7 s",16,"\u0001","minmax","GPT-6.1 Sol · effort high · eight hard validated tasks",[8.93,75.91],"\u0001","\u0001","\u0001"],["hard-model-head-to-head","hard-h2h-output-tokens","Output tokens","GPT-6.1 Sol (medium) · Codex CLI","hard-h2h-output-tokens/Output tokens","Output tokens per call on hard tasks (Output tokens)",335,"tokens","335",16,"\u0001","\u0001","GPT-6.1 Sol · effort medium · eight hard validated tasks","\u0001","\u0001","\u0001","\u0001"],["hard-model-head-to-head","hard-h2h-output-tokens","Output tokens","GPT-6.1 Sol (high) · Codex CLI","hard-h2h-output-tokens/Output tokens","Output tokens per call on hard tasks (Output tokens)",436,"tokens","436",16,"\u0001","\u0001","GPT-6.1 Sol · effort high · eight hard validated tasks","\u0001","\u0001","\u0001","\u0001"],["hard-model-head-to-head","hard-h2h-cost-per-pass","Cost per strict pass","GPT-6.1 Sol (high) · Codex CLI","hard-h2h-cost-per-pass","List-price cost per strict pass on hard tasks (calculation)",0.01514,"usd","$0.015",16,"\u0001","\u0001","GPT-6.1 Sol · effort high · eight hard validated tasks","\u0001",true,"\u0001","\u0001"],["hard-model-head-to-head","hard-h2h-cost-per-pass","Cost per strict pass","GPT-6.1 Sol (medium) · Codex CLI","hard-h2h-cost-per-pass","List-price cost per strict pass on hard tasks (calculation)",0.02564,"usd","$0.026",16,"\u0001","\u0001","GPT-6.1 Sol · effort medium · eight hard validated tasks","\u0001",true,"\u0001","\u0001"],["coding-agents-head-to-head","coding-agents-pass-rate","Passed every hidden check","GPT-6.1 Sol (medium, tester’s AGENTS.md) · Codex CLI","coding-agents-pass-rate","Coding sessions that passed every hidden check",1,"rate","100% (12/12)",12,[0.7575,1],"ci95","GPT-6.1 Sol · effort medium · tester’s AGENTS.md · six small repository tasks with hidden tests","\u0001","\u0001","\u0001","\u0001"],["coding-agents-head-to-head","coding-agents-wall-time","Wall time per session","GPT-6.1 Sol (medium, tester’s AGENTS.md) · Codex CLI","coding-agents-wall-time","Time per coding session",113.4,"seconds","113.4 s",12,"\u0001","minmax","GPT-6.1 Sol · effort medium · tester’s AGENTS.md · six small repository tasks with hidden tests",[78.5,221.9],"\u0001","\u0001","\u0001"],["coding-agents-head-to-head","coding-agents-tool-calls","Tool calls per session","GPT-6.1 Sol (medium, tester’s AGENTS.md) · Codex CLI","coding-agents-tool-calls","Tool calls per coding session",12.5,"calls","12.5",12,"\u0001","minmax","GPT-6.1 Sol · effort medium · tester’s AGENTS.md · six small repository tasks with hidden tests",[8,18],"\u0001","\u0001","\u0001"],["coding-agents-head-to-head","coding-agents-cost-per-pass","List-price cost per pass","GPT-6.1 Sol (medium, tester’s AGENTS.md) · Codex CLI","coding-agents-cost-per-pass","List-price cost per passing coding session (calculation)",0.0978,"usd","$0.098",12,"\u0001","\u0001","GPT-6.1 Sol · effort medium · tester’s AGENTS.md · six small repository tasks with hidden tests","\u0001",true,"\u0001","\u0001"],["effort-ladder","effort-ladder-pass-rate","Strict pass","GPT-6.1 Sol (low) · Codex CLI","effort-ladder-pass-rate","Strict pass rate by effort on eight hard tasks",1,"rate","100% (16/16)",16,[0.8064,1],"ci95","GPT-6.1 Sol · effort low · eight hard validated tasks, effort ladder","\u0001","\u0001","\u0001","\u0001"],["effort-ladder","effort-ladder-pass-rate","Strict pass","GPT-6.1 Sol (medium) · Codex CLI","effort-ladder-pass-rate","Strict pass rate by effort on eight hard tasks",1,"rate","100% (16/16)",16,[0.8064,1],"ci95","GPT-6.1 Sol · effort medium · eight hard validated tasks, effort ladder","\u0001","\u0001","\u0001","\u0001"],["effort-ladder","effort-ladder-pass-rate","Strict pass","GPT-6.1 Sol (high) · Codex CLI","effort-ladder-pass-rate","Strict pass rate by effort on eight hard tasks",1,"rate","100% (16/16)",16,[0.8064,1],"ci95","GPT-6.1 Sol · effort high · eight hard validated tasks, effort ladder","\u0001","\u0001","\u0001","\u0001"],["effort-ladder","effort-ladder-total-latency","Total time per call by effort on hard tasks","GPT-6.1 Sol (low) · Codex CLI","effort-ladder-total-latency","Total time per call by effort on hard tasks",13.62,"seconds","13.6 s",16,"\u0001","minmax","GPT-6.1 Sol · effort low · eight hard validated tasks, effort ladder",[7.94,44.29],"\u0001","\u0001","\u0001"],["effort-ladder","effort-ladder-total-latency","Total time per call by effort on hard tasks","GPT-6.1 Sol (medium) · Codex CLI","effort-ladder-total-latency","Total time per call by effort on hard tasks",13.11,"seconds","13.1 s",16,"\u0001","minmax","GPT-6.1 Sol · effort medium · eight hard validated tasks, effort ladder",[8.54,61.6],"\u0001","\u0001","\u0001"],["effort-ladder","effort-ladder-total-latency","Total time per call by effort on hard tasks","GPT-6.1 Sol (high) · Codex CLI","effort-ladder-total-latency","Total time per call by effort on hard tasks",18.12,"seconds","18.1 s",16,"\u0001","minmax","GPT-6.1 Sol · effort high · eight hard validated tasks, effort ladder",[11.67,92.21],"\u0001","\u0001","\u0001"],["effort-ladder","effort-ladder-output-tokens","Output tokens","GPT-6.1 Sol (low) · Codex CLI","effort-ladder-output-tokens/Output tokens","Output tokens per call by effort on hard tasks (Output tokens)",284,"tokens","284",16,"\u0001","\u0001","GPT-6.1 Sol · effort low · eight hard validated tasks, effort ladder","\u0001","\u0001","\u0001","\u0001"],["effort-ladder","effort-ladder-output-tokens","Output tokens","GPT-6.1 Sol (medium) · Codex CLI","effort-ladder-output-tokens/Output tokens","Output tokens per call by effort on hard tasks (Output tokens)",335,"tokens","335",16,"\u0001","\u0001","GPT-6.1 Sol · effort medium · eight hard validated tasks, effort ladder","\u0001","\u0001","\u0001","\u0001"],["effort-ladder","effort-ladder-output-tokens","Output tokens","GPT-6.1 Sol (high) · Codex CLI","effort-ladder-output-tokens/Output tokens","Output tokens per call by effort on hard tasks (Output tokens)",436,"tokens","436",16,"\u0001","\u0001","GPT-6.1 Sol · effort high · eight hard validated tasks, effort ladder","\u0001","\u0001","\u0001","\u0001"],["effort-ladder","effort-ladder-cost-per-pass","Cost per strict pass","GPT-6.1 Sol (low) · Codex CLI","effort-ladder-cost-per-pass","List-price cost per strict pass by effort (calculation)",0.01284,"usd","$0.013",16,"\u0001","\u0001","GPT-6.1 Sol · effort low · eight hard validated tasks, effort ladder","\u0001",true,"\u0001","\u0001"],["effort-ladder","effort-ladder-cost-per-pass","Cost per strict pass","GPT-6.1 Sol (medium) · Codex CLI","effort-ladder-cost-per-pass","List-price cost per strict pass by effort (calculation)",0.02564,"usd","$0.026",16,"\u0001","\u0001","GPT-6.1 Sol · effort medium · eight hard validated tasks, effort ladder","\u0001",true,"\u0001","\u0001"],["effort-ladder","effort-ladder-cost-per-pass","Cost per strict pass","GPT-6.1 Sol (high) · Codex CLI","effort-ladder-cost-per-pass","List-price cost per strict pass by effort (calculation)",0.01514,"usd","$0.015",16,"\u0001","\u0001","GPT-6.1 Sol · effort high · eight hard validated tasks, effort ladder","\u0001",true,"\u0001","\u0001"],["caching-consistency","consistency-pass-rate","Exact number","GPT-6.1 Sol (medium) · Codex CLI","consistency-pass-rate/Exact number","Same prompt, 10 times: strict pass rate (Exact number)",1,"rate","100% (10/10)",10,[0.7225,1],"ci95","GPT-6.1 Sol · effort medium · same prompt repeated 10 times","\u0001","\u0001","\u0001","\u0001"],["caching-consistency","consistency-pass-rate","JSON object","GPT-6.1 Sol (medium) · Codex CLI","consistency-pass-rate/JSON object","Same prompt, 10 times: strict pass rate (JSON object)",1,"rate","100% (10/10)",10,[0.7225,1],"ci95","GPT-6.1 Sol · effort medium · same prompt repeated 10 times","\u0001","\u0001","\u0001","\u0001"],["caching-consistency","consistency-pass-rate","Code fix","GPT-6.1 Sol (medium) · Codex CLI","consistency-pass-rate/Code fix","Same prompt, 10 times: strict pass rate (Code fix)",1,"rate","100% (10/10)",10,[0.7225,1],"ci95","GPT-6.1 Sol · effort medium · same prompt repeated 10 times","\u0001","\u0001","\u0001","\u0001"],["caching-consistency","consistency-distinct-answers","Exact number","GPT-6.1 Sol (medium) · Codex CLI","consistency-distinct-answers/Exact number","Same prompt, 10 times: how many different answers (Exact number)",1,"count","1",10,"\u0001","\u0001","GPT-6.1 Sol · effort medium · same prompt repeated 10 times","\u0001","\u0001","\u0001","\u0001"],["caching-consistency","consistency-distinct-answers","JSON object","GPT-6.1 Sol (medium) · Codex CLI","consistency-distinct-answers/JSON object","Same prompt, 10 times: how many different answers (JSON object)",1,"count","1",10,"\u0001","\u0001","GPT-6.1 Sol · effort medium · same prompt repeated 10 times","\u0001","\u0001","\u0001","\u0001"],["caching-consistency","consistency-distinct-answers","Code fix","GPT-6.1 Sol (medium) · Codex CLI","consistency-distinct-answers/Code fix","Same prompt, 10 times: how many different answers (Code fix)",6,"count","6",10,"\u0001","\u0001","GPT-6.1 Sol · effort medium · same prompt repeated 10 times","\u0001","\u0001","\u0001","\u0001"],["caching-consistency","consistency-latency-spread","Exact number","GPT-6.1 Sol (medium) · Codex CLI","consistency-latency-spread/Exact number","Same prompt, 10 times: time per call (Exact number)",13.38,"seconds","13.4 s",10,"\u0001","minmax","GPT-6.1 Sol · effort medium · same prompt repeated 10 times",[12.29,17.97],"\u0001","\u0001","\u0001"],["caching-consistency","consistency-latency-spread","JSON object","GPT-6.1 Sol (medium) · Codex CLI","consistency-latency-spread/JSON object","Same prompt, 10 times: time per call (JSON object)",6.42,"seconds","6.42 s",10,"\u0001","minmax","GPT-6.1 Sol · effort medium · same prompt repeated 10 times",[5.25,8.26],"\u0001","\u0001","\u0001"],["caching-consistency","consistency-latency-spread","Code fix","GPT-6.1 Sol (medium) · Codex CLI","consistency-latency-spread/Code fix","Same prompt, 10 times: time per call (Code fix)",11.29,"seconds","11.3 s",10,"\u0001","minmax","GPT-6.1 Sol · effort medium · same prompt repeated 10 times",[9.08,14.85],"\u0001","\u0001","\u0001"],["routing-overhead","cli-startup-tax","First output event","Codex CLI (default model)","cli-startup-tax/First output event","CLI start-up tax on a one-word answer (First output event)",489,"ms","489 ms",5,"\u0001","minmax","default model · CLI start-up, one-word prompt, 5 runs",[354,1304],"\u0001","\u0001","\u0001"],["routing-overhead","cli-startup-tax","First model output","Codex CLI (default model)","cli-startup-tax/First model output","CLI start-up tax on a one-word answer (First model output)",5059,"ms","5,059 ms",5,"\u0001","minmax","default model · CLI start-up, one-word prompt, 5 runs",[4391,5478],"\u0001","\u0001","\u0001"],["routing-overhead","cli-startup-tax","Total wall time","Codex CLI (default model)","cli-startup-tax/Total wall time","CLI start-up tax on a one-word answer (Total wall time)",5999,"ms","5,999 ms",5,"\u0001","minmax","default model · CLI start-up, one-word prompt, 5 runs",[5367,6506],"\u0001","\u0001","\u0001"],["routing-overhead","cli-startup-input-tokens","Input tokens per call","Codex CLI (default model)","cli-startup-input-tokens","Input tokens a CLI sends for a one-word answer",17051,"tokens","17,051",5,"\u0001","\u0001","default model · CLI start-up, one-word prompt, 5 runs","\u0001","\u0001","\u0001","\u0001"],["cli-model-latency-tokens","cli-vs-api-exact-reply-latency","Total time","Codex CLI · GPT-6 Luna · none","cli-vs-api-exact-reply-latency/Total time","CLI vs API: time for a one-line answer (Total time)",3.19,"seconds","3.19 s",5,"\u0001","minmax","GPT-6 Luna · effort none · fixed exact reply, 5 runs",[2.88,3.83],"\u0001","\u0001","\u0001"],["cli-model-latency-tokens","cli-vs-api-exact-reply-latency","Total time","Codex CLI · GPT-6.1 Sol · low","cli-vs-api-exact-reply-latency/Total time","CLI vs API: time for a one-line answer (Total time)",4.18,"seconds","4.18 s",5,"\u0001","minmax","GPT-6.1 Sol · effort low · fixed exact reply, 5 runs",[3.86,4.53],"\u0001","\u0001","\u0001"],["cli-model-latency-tokens","cli-vs-api-exact-reply-latency","Total time","Codex CLI · GPT-6.1 Sol · high","cli-vs-api-exact-reply-latency/Total time","CLI vs API: time for a one-line answer (Total time)",4.19,"seconds","4.19 s",5,"\u0001","minmax","GPT-6.1 Sol · effort high · fixed exact reply, 5 runs",[3.81,4.69],"\u0001","\u0001","\u0001"],["cli-model-latency-tokens","cli-vs-api-exact-reply-latency","First useful output","Codex CLI · GPT-6 Luna · none","cli-vs-api-exact-reply-latency/First useful output","CLI vs API: time for a one-line answer (First useful output)",2.79,"seconds","2.79 s",5,"\u0001","minmax","GPT-6 Luna · effort none · fixed exact reply, 5 runs",[2.46,3.42],"\u0001","\u0001","\u0001"],["cli-model-latency-tokens","cli-vs-api-exact-reply-latency","First useful output","Codex CLI · GPT-6.1 Sol · low","cli-vs-api-exact-reply-latency/First useful output","CLI vs API: time for a one-line answer (First useful output)",3.75,"seconds","3.75 s",5,"\u0001","minmax","GPT-6.1 Sol · effort low · fixed exact reply, 5 runs",[3.44,4.1],"\u0001","\u0001","\u0001"],["cli-model-latency-tokens","cli-vs-api-exact-reply-latency","First useful output","Codex CLI · GPT-6.1 Sol · high","cli-vs-api-exact-reply-latency/First useful output","CLI vs API: time for a one-line answer (First useful output)",3.79,"seconds","3.79 s",5,"\u0001","minmax","GPT-6.1 Sol · effort high · fixed exact reply, 5 runs",[3.37,4.3],"\u0001","\u0001","\u0001"],["cli-model-latency-tokens","cli-vs-api-small-coding-latency","Total time","Codex CLI · GPT-6 Luna · none","cli-vs-api-small-coding-latency/Total time","CLI vs API: time for a small coding task (Total time)",9.23,"seconds","9.23 s",3,"\u0001","minmax","GPT-6 Luna · effort none · small coding task, 3 runs",[8.99,11.68],"\u0001","\u0001","\u0001"],["cli-model-latency-tokens","cli-vs-api-small-coding-latency","Total time","Codex CLI · GPT-6.1 Sol · low","cli-vs-api-small-coding-latency/Total time","CLI vs API: time for a small coding task (Total time)",14.15,"seconds","14.2 s",3,"\u0001","minmax","GPT-6.1 Sol · effort low · small coding task, 3 runs",[13.02,14.41],"\u0001","\u0001","\u0001"],["cli-model-latency-tokens","cli-vs-api-small-coding-latency","Total time","Codex CLI · GPT-6.1 Sol · high","cli-vs-api-small-coding-latency/Total time","CLI vs API: time for a small coding task (Total time)",17.85,"seconds","17.9 s",3,"\u0001","minmax","GPT-6.1 Sol · effort high · small coding task, 3 runs",[17.68,22.42],"\u0001","\u0001","\u0001"],["cli-model-latency-tokens","cli-vs-api-small-coding-latency","First useful output","Codex CLI · GPT-6 Luna · none","cli-vs-api-small-coding-latency/First useful output","CLI vs API: time for a small coding task (First useful output)",8.68,"seconds","8.68 s",3,"\u0001","minmax","GPT-6 Luna · effort none · small coding task, 3 runs",[8.27,11.01],"\u0001","\u0001","\u0001"],["cli-model-latency-tokens","cli-vs-api-small-coding-latency","First useful output","Codex CLI · GPT-6.1 Sol · low","cli-vs-api-small-coding-latency/First useful output","CLI vs API: time for a small coding task (First useful output)",13.6,"seconds","13.6 s",3,"\u0001","minmax","GPT-6.1 Sol · effort low · small coding task, 3 runs",[12.52,13.83],"\u0001","\u0001","\u0001"],["cli-model-latency-tokens","cli-vs-api-small-coding-latency","First useful output","Codex CLI · GPT-6.1 Sol · high","cli-vs-api-small-coding-latency/First useful output","CLI vs API: time for a small coding task (First useful output)",17.27,"seconds","17.3 s",3,"\u0001","minmax","GPT-6.1 Sol · effort high · small coding task, 3 runs",[17.13,21.86],"\u0001","\u0001","\u0001"],["cli-model-latency-tokens","cli-vs-api-prompt-overhead","Input tokens","Codex CLI · GPT-6 Luna · none","cli-vs-api-prompt-overhead","Hidden prompt: input tokens for the same one-line request",18859,"tokens","18,859",5,"\u0001","\u0001","GPT-6 Luna · effort none · short fixed tasks","\u0001","\u0001","\u0001","\u0001"],["cli-model-latency-tokens","cli-vs-api-prompt-overhead","Input tokens","Codex CLI · GPT-6.1 Sol · low","cli-vs-api-prompt-overhead","Hidden prompt: input tokens for the same one-line request",19551,"tokens","19,551",5,"\u0001","\u0001","GPT-6.1 Sol · effort low · short fixed tasks","\u0001","\u0001","\u0001","\u0001"],["cli-model-latency-tokens","cli-vs-api-prompt-overhead","Input tokens","Codex CLI · GPT-6.1 Sol · high","cli-vs-api-prompt-overhead","Hidden prompt: input tokens for the same one-line request",19555,"tokens","19,555",5,"\u0001","\u0001","GPT-6.1 Sol · effort high · short fixed tasks","\u0001","\u0001","\u0001","\u0001"],["cli-model-latency-tokens","scheduler-repair-claude-vs-codex","Total time","Codex CLI · GPT-6.1 Sol · medium","scheduler-repair-claude-vs-codex/Total time","Repairing a scheduler: Claude Code vs Codex vs API (Total time)",61.16,"seconds","61.2 s",3,"\u0001","minmax","GPT-6.1 Sol · effort medium · scheduler repair, 296 checks, 3 runs",[59.9,69.51],"\u0001","\u0001","\u0001"],["cli-model-latency-tokens","scheduler-repair-claude-vs-codex","First useful output","Codex CLI · GPT-6.1 Sol · medium","scheduler-repair-claude-vs-codex/First useful output","Repairing a scheduler: Claude Code vs Codex vs API (First useful output)",15.56,"seconds","15.6 s",3,"\u0001","minmax","GPT-6.1 Sol · effort medium · scheduler repair, 296 checks, 3 runs",[13.65,23.04],"\u0001","\u0001","\u0001"],["cli-model-latency-tokens","scheduler-repair-output-tokens","Output tokens","Codex CLI · GPT-6.1 Sol · medium","scheduler-repair-output-tokens/Output tokens","Output tokens to repair the scheduler (Output tokens)",1181,"tokens","1,181",3,"\u0001","\u0001","GPT-6.1 Sol · effort medium · scheduler repair, 296 checks, 3 runs","\u0001","\u0001","\u0001","\u0001"],["cli-model-latency-tokens","\u0001","\u0001","\u0001","stat:exact-reply-cli-over-api","Codex CLI vs OpenAI API, median total time for a one-line answer",3.49,"ratio","3.5x slower",30,"\u0001","\u0001","short fixed tasks","\u0001","\u0001","exact-reply-cli-over-api","\u0001"],["single-call-vs-agent-loop","agent-loop-pass-rate","Strict pass","GPT-6 Luna (single call) · Codex CLI","agent-loop-pass-rate","Strict pass rate: single call vs agent loop on eight hard tasks",0.625,"rate","63% (10/16)",16,[0.3864,0.8152],"ci95","GPT-6 Luna · single call","\u0001","\u0001","\u0001","higher"],["single-call-vs-agent-loop","agent-loop-pass-rate","Strict pass","GPT-6 Luna (agent loop) · Codex CLI","agent-loop-pass-rate","Strict pass rate: single call vs agent loop on eight hard tasks",0.8571,"rate","86% (12/14)",14,[0.6006,0.9599],"ci95","GPT-6 Luna · agent loop","\u0001","\u0001","\u0001","higher"],["single-call-vs-agent-loop","agent-loop-by-task","GPT-6 Luna (single call) · Codex CLI","Interval merge fix","agent-loop-by-task/Interval merge fix","Strict passes per task: single call vs agent loop: Interval merge fix",1,"rate","100% (2/2)",2,[0.3424,1],"ci95","GPT-6 Luna · single call","\u0001","\u0001","\u0001","higher"],["single-call-vs-agent-loop","agent-loop-by-task","GPT-6 Luna (single call) · Codex CLI","DST day-length fix","agent-loop-by-task/DST day-length fix","Strict passes per task: single call vs agent loop: DST day-length fix",1,"rate","100% (2/2)",2,[0.3424,1],"ci95","GPT-6 Luna · single call","\u0001","\u0001","\u0001","higher"],["single-call-vs-agent-loop","agent-loop-by-task","GPT-6 Luna (single call) · Codex CLI","CSV parser","agent-loop-by-task/CSV parser","Strict passes per task: single call vs agent loop: CSV parser",1,"rate","100% (2/2)",2,[0.3424,1],"ci95","GPT-6 Luna · single call","\u0001","\u0001","\u0001","higher"],["single-call-vs-agent-loop","agent-loop-by-task","GPT-6 Luna (single call) · Codex CLI","Event-loop order","agent-loop-by-task/Event-loop order","Strict passes per task: single call vs agent loop: Event-loop order",0,"rate","0% (0/2)",2,[0,0.6576],"ci95","GPT-6 Luna · single call","\u0001","\u0001","\u0001","higher"],["single-call-vs-agent-loop","agent-loop-by-task","GPT-6 Luna (single call) · Codex CLI","Room schedule","agent-loop-by-task/Room schedule","Strict passes per task: single call vs agent loop: Room schedule",0.5,"rate","50% (1/2)",2,[0.0945,0.9055],"ci95","GPT-6 Luna · single call","\u0001","\u0001","\u0001","higher"],["single-call-vs-agent-loop","agent-loop-by-task","GPT-6 Luna (single call) · Codex CLI","SemVer regex","agent-loop-by-task/SemVer regex","Strict passes per task: single call vs agent loop: SemVer regex",0.5,"rate","50% (1/2)",2,[0.0945,0.9055],"ci95","GPT-6 Luna · single call","\u0001","\u0001","\u0001","higher"],["single-call-vs-agent-loop","agent-loop-by-task","GPT-6 Luna (single call) · Codex CLI","Money refactor","agent-loop-by-task/Money refactor","Strict passes per task: single call vs agent loop: Money refactor",0,"rate","0% (0/2)",2,[0,0.6576],"ci95","GPT-6 Luna · single call","\u0001","\u0001","\u0001","higher"],["single-call-vs-agent-loop","agent-loop-by-task","GPT-6 Luna (single call) · Codex CLI","SQL report","agent-loop-by-task/SQL report","Strict passes per task: single call vs agent loop: SQL report",1,"rate","100% (2/2)",2,[0.3424,1],"ci95","GPT-6 Luna · single call","\u0001","\u0001","\u0001","higher"],["single-call-vs-agent-loop","agent-loop-by-task","GPT-6 Luna (agent loop) · Codex CLI","Interval merge fix","agent-loop-by-task/Interval merge fix","Strict passes per task: single call vs agent loop: Interval merge fix",1,"rate","100% (2/2)",2,[0.3424,1],"ci95","GPT-6 Luna · agent loop","\u0001","\u0001","\u0001","higher"],["single-call-vs-agent-loop","agent-loop-by-task","GPT-6 Luna (agent loop) · Codex CLI","CSV parser","agent-loop-by-task/CSV parser","Strict passes per task: single call vs agent loop: CSV parser",0.5,"rate","50% (1/2)",2,[0.0945,0.9055],"ci95","GPT-6 Luna · agent loop","\u0001","\u0001","\u0001","higher"],["single-call-vs-agent-loop","agent-loop-by-task","GPT-6 Luna (agent loop) · Codex CLI","Event-loop order","agent-loop-by-task/Event-loop order","Strict passes per task: single call vs agent loop: Event-loop order",1,"rate","100% (2/2)",2,[0.3424,1],"ci95","GPT-6 Luna · agent loop","\u0001","\u0001","\u0001","higher"],["single-call-vs-agent-loop","agent-loop-by-task","GPT-6 Luna (agent loop) · Codex CLI","Room schedule","agent-loop-by-task/Room schedule","Strict passes per task: single call vs agent loop: Room schedule",1,"rate","100% (2/2)",2,[0.3424,1],"ci95","GPT-6 Luna · agent loop","\u0001","\u0001","\u0001","higher"],["single-call-vs-agent-loop","agent-loop-by-task","GPT-6 Luna (agent loop) · Codex CLI","SemVer regex","agent-loop-by-task/SemVer regex","Strict passes per task: single call vs agent loop: SemVer regex",1,"rate","100% (2/2)",2,[0.3424,1],"ci95","GPT-6 Luna · agent loop","\u0001","\u0001","\u0001","higher"],["single-call-vs-agent-loop","agent-loop-by-task","GPT-6 Luna (agent loop) · Codex CLI","Money refactor","agent-loop-by-task/Money refactor","Strict passes per task: single call vs agent loop: Money refactor",0.5,"rate","50% (1/2)",2,[0.0945,0.9055],"ci95","GPT-6 Luna · agent loop","\u0001","\u0001","\u0001","higher"],["single-call-vs-agent-loop","agent-loop-by-task","GPT-6 Luna (agent loop) · Codex CLI","SQL report","agent-loop-by-task/SQL report","Strict passes per task: single call vs agent loop: SQL report",1,"rate","100% (2/2)",2,[0.3424,1],"ci95","GPT-6 Luna · agent loop","\u0001","\u0001","\u0001","higher"],["single-call-vs-agent-loop","agent-loop-total-time","Total time per attempt","GPT-6 Luna (single call) · Codex CLI","agent-loop-total-time","Total time per attempt: single call vs agent loop",5.16,"seconds","5.16 s",16,"\u0001","minmax","GPT-6 Luna · single call",[3.59,11.32],"\u0001","\u0001","\u0001"],["single-call-vs-agent-loop","agent-loop-total-time","Total time per attempt","GPT-6 Luna (agent loop) · Codex CLI","agent-loop-total-time","Total time per attempt: single call vs agent loop",9.32,"seconds","9.32 s",14,"\u0001","minmax","GPT-6 Luna · agent loop",[3.78,15.89],"\u0001","\u0001","\u0001"],["single-call-vs-agent-loop","agent-loop-tokens","Input tokens (cache reads included)","GPT-6 Luna (single call) · Codex CLI","agent-loop-tokens/Input tokens (cache reads included)","Tokens per attempt: single call vs agent loop (Input tokens (cache reads included))",11582,"tokens","11,582",16,"\u0001","minmax","GPT-6 Luna · single call",[11526,11818],"\u0001","\u0001","\u0001"],["single-call-vs-agent-loop","agent-loop-tokens","Input tokens (cache reads included)","GPT-6 Luna (agent loop) · Codex CLI","agent-loop-tokens/Input tokens (cache reads included)","Tokens per attempt: single call vs agent loop (Input tokens (cache reads included))",15530,"tokens","15,530",14,"\u0001","minmax","GPT-6 Luna · agent loop",[15391,39009],"\u0001","\u0001","\u0001"],["single-call-vs-agent-loop","agent-loop-tokens","Output tokens","GPT-6 Luna (single call) · Codex CLI","agent-loop-tokens/Output tokens","Tokens per attempt: single call vs agent loop (Output tokens)",345,"tokens","345",16,"\u0001","minmax","GPT-6 Luna · single call",[36,634],"\u0001","\u0001","\u0001"],["single-call-vs-agent-loop","agent-loop-tokens","Output tokens","GPT-6 Luna (agent loop) · Codex CLI","agent-loop-tokens/Output tokens","Tokens per attempt: single call vs agent loop (Output tokens)",480,"tokens","480",14,"\u0001","minmax","GPT-6 Luna · agent loop",[143,858],"\u0001","\u0001","\u0001"],["single-call-vs-agent-loop","agent-loop-tool-calls","Tool calls per attempt","GPT-6 Luna (agent loop) · Codex CLI","agent-loop-tool-calls","Tool calls per agent-loop attempt",0,"count","0",14,"\u0001","minmax","GPT-6 Luna · agent loop",[0,1],"\u0001","\u0001","\u0001"],["single-call-vs-agent-loop","agent-loop-cost-per-pass","Cost per strict pass","GPT-6 Luna (single call) · Codex CLI","agent-loop-cost-per-pass","List-price cost per strict pass: single call vs agent loop (calculation)",0.00116,"usd","$0.0012",16,"\u0001","\u0001","GPT-6 Luna · single call","\u0001",true,"\u0001","\u0001"],["single-call-vs-agent-loop","agent-loop-cost-per-pass","Cost per strict pass","GPT-6 Luna (agent loop) · Codex CLI","agent-loop-cost-per-pass","List-price cost per strict pass: single call vs agent loop (calculation)",0.00099,"usd","$0.00099",14,"\u0001","\u0001","GPT-6 Luna · agent loop","\u0001",true,"\u0001","\u0001"],["json-schema-vs-instructions","structured-output-pass-rate","Strict pass: the whole reply is the right JSON","GPT-6.1 Sol (low, instructions) · Codex CLI","structured-output-pass-rate/Strict pass: the whole reply is the right JSON","Does a JSON schema raise the pass rate? Instructions vs schema mode (Strict pass: the whole reply is the right JSON)",1,"rate","100% (12/12)",12,[0.7575,1],"ci95","GPT-6.1 Sol · effort low · instructions","\u0001",true,"\u0001","higher"],["json-schema-vs-instructions","structured-output-pass-rate","Strict pass: the whole reply is the right JSON","GPT-6.1 Sol (low, JSON schema) · Codex CLI","structured-output-pass-rate/Strict pass: the whole reply is the right JSON","Does a JSON schema raise the pass rate? Instructions vs schema mode (Strict pass: the whole reply is the right JSON)",1,"rate","100% (12/12)",12,[0.7575,1],"ci95","GPT-6.1 Sol · effort low · JSON schema","\u0001",true,"\u0001","higher"],["json-schema-vs-instructions","structured-output-pass-rate","Right answer in any format (strict pass or format miss)","GPT-6.1 Sol (low, instructions) · Codex CLI","structured-output-pass-rate/Right answer in any format (strict pass or format miss)","Does a JSON schema raise the pass rate? Instructions vs schema mode (Right answer in any format (strict pass or format miss))",1,"rate","100% (12/12)",12,[0.7575,1],"ci95","GPT-6.1 Sol · effort low · instructions","\u0001",true,"\u0001","higher"],["json-schema-vs-instructions","structured-output-pass-rate","Right answer in any format (strict pass or format miss)","GPT-6.1 Sol (low, JSON schema) · Codex CLI","structured-output-pass-rate/Right answer in any format (strict pass or format miss)","Does a JSON schema raise the pass rate? Instructions vs schema mode (Right answer in any format (strict pass or format miss))",1,"rate","100% (12/12)",12,[0.7575,1],"ci95","GPT-6.1 Sol · effort low · JSON schema","\u0001",true,"\u0001","higher"],["json-schema-vs-instructions","structured-output-outcomes","Strict pass","GPT-6.1 Sol (low, instructions) · Codex CLI","structured-output-outcomes/Strict pass","What each call produced: strict pass, format miss, wrong values or error (Strict pass)",12,"count","12",12,"\u0001","\u0001","GPT-6.1 Sol · effort low · instructions","\u0001","\u0001","\u0001","\u0001"],["json-schema-vs-instructions","structured-output-outcomes","Strict pass","GPT-6.1 Sol (low, JSON schema) · Codex CLI","structured-output-outcomes/Strict pass","What each call produced: strict pass, format miss, wrong values or error (Strict pass)",12,"count","12",12,"\u0001","\u0001","GPT-6.1 Sol · effort low · JSON schema","\u0001","\u0001","\u0001","\u0001"],["json-schema-vs-instructions","structured-output-outcomes","Format miss","GPT-6.1 Sol (low, instructions) · Codex CLI","structured-output-outcomes/Format miss","What each call produced: strict pass, format miss, wrong values or error (Format miss)",0,"count","0",12,"\u0001","\u0001","GPT-6.1 Sol · effort low · instructions","\u0001","\u0001","\u0001","\u0001"],["json-schema-vs-instructions","structured-output-outcomes","Format miss","GPT-6.1 Sol (low, JSON schema) · Codex CLI","structured-output-outcomes/Format miss","What each call produced: strict pass, format miss, wrong values or error (Format miss)",0,"count","0",12,"\u0001","\u0001","GPT-6.1 Sol · effort low · JSON schema","\u0001","\u0001","\u0001","\u0001"],["json-schema-vs-instructions","structured-output-outcomes","Wrong values","GPT-6.1 Sol (low, instructions) · Codex CLI","structured-output-outcomes/Wrong values","What each call produced: strict pass, format miss, wrong values or error (Wrong values)",0,"count","0",12,"\u0001","\u0001","GPT-6.1 Sol · effort low · instructions","\u0001","\u0001","\u0001","\u0001"],["json-schema-vs-instructions","structured-output-outcomes","Wrong values","GPT-6.1 Sol (low, JSON schema) · Codex CLI","structured-output-outcomes/Wrong values","What each call produced: strict pass, format miss, wrong values or error (Wrong values)",0,"count","0",12,"\u0001","\u0001","GPT-6.1 Sol · effort low · JSON schema","\u0001","\u0001","\u0001","\u0001"],["json-schema-vs-instructions","structured-output-outcomes","Error","GPT-6.1 Sol (low, instructions) · Codex CLI","structured-output-outcomes/Error","What each call produced: strict pass, format miss, wrong values or error (Error)",0,"count","0",12,"\u0001","\u0001","GPT-6.1 Sol · effort low · instructions","\u0001","\u0001","\u0001","\u0001"],["json-schema-vs-instructions","structured-output-outcomes","Error","GPT-6.1 Sol (low, JSON schema) · Codex CLI","structured-output-outcomes/Error","What each call produced: strict pass, format miss, wrong values or error (Error)",0,"count","0",12,"\u0001","\u0001","GPT-6.1 Sol · effort low · JSON schema","\u0001","\u0001","\u0001","\u0001"],["json-schema-vs-instructions","structured-output-time","Median time per call (the three prompts pooled)","GPT-6.1 Sol (low, instructions) · Codex CLI","structured-output-time","Time per call, instructions vs schema mode",6.21,"seconds","6.21 s",12,"\u0001","minmax","GPT-6.1 Sol · effort low · instructions",[4.2,12.27],"\u0001","\u0001","\u0001"],["json-schema-vs-instructions","structured-output-time","Median time per call (the three prompts pooled)","GPT-6.1 Sol (low, JSON schema) · Codex CLI","structured-output-time","Time per call, instructions vs schema mode",5.96,"seconds","5.96 s",12,"\u0001","minmax","GPT-6.1 Sol · effort low · JSON schema",[4.62,20.97],"\u0001","\u0001","\u0001"],["json-schema-vs-instructions","structured-output-tokens","Median output tokens per call","GPT-6.1 Sol (low, instructions) · Codex CLI","structured-output-tokens/Median output tokens per call","Output and reasoning tokens per call, instructions vs schema mode (Median output tokens per call)",117,"tokens","117",12,"\u0001","\u0001","GPT-6.1 Sol · effort low · instructions","\u0001","\u0001","\u0001","\u0001"],["json-schema-vs-instructions","structured-output-tokens","Median output tokens per call","GPT-6.1 Sol (low, JSON schema) · Codex CLI","structured-output-tokens/Median output tokens per call","Output and reasoning tokens per call, instructions vs schema mode (Median output tokens per call)",123,"tokens","123",12,"\u0001","\u0001","GPT-6.1 Sol · effort low · JSON schema","\u0001","\u0001","\u0001","\u0001"],["thinking-token-bill","thinking-bill-share","Median call","GPT-6.1 Sol (high) · Codex CLI","thinking-bill-share","Reasoning share of output tokens per call on hard tasks (calculation)",57.01,"percent","57%",16,"\u0001","minmax","GPT-6.1 Sol · effort high",[29.19,90.8],true,"\u0001","none"],["thinking-token-bill","thinking-bill-share","Median call","GPT-6.1 Sol (medium) · Codex CLI","thinking-bill-share","Reasoning share of output tokens per call on hard tasks (calculation)",46.33,"percent","46.3%",16,"\u0001","minmax","GPT-6.1 Sol · effort medium",[11.42,86.85],true,"\u0001","none"],["thinking-token-bill","thinking-bill-cost-per-call","Reasoning (output tokens)","GPT-6.1 Sol (medium) · Codex CLI","thinking-bill-cost-per-call/Reasoning (output tokens)","List-price cost per call: reasoning, remaining output and input (calculation) (Reasoning (output tokens))",0.002273,"usd","$0.0023",16,"\u0001","\u0001","GPT-6.1 Sol · effort medium","\u0001",true,"\u0001","none"],["thinking-token-bill","thinking-bill-cost-per-call","Reasoning (output tokens)","GPT-6.1 Sol (high) · Codex CLI","thinking-bill-cost-per-call/Reasoning (output tokens)","List-price cost per call: reasoning, remaining output and input (calculation) (Reasoning (output tokens))",0.004114,"usd","$0.0041",16,"\u0001","\u0001","GPT-6.1 Sol · effort high","\u0001",true,"\u0001","none"],["thinking-token-bill","thinking-bill-cost-per-call","Remaining output (visible-answer estimate)","GPT-6.1 Sol (medium) · Codex CLI","thinking-bill-cost-per-call/Remaining output (visible-answer estimate)","List-price cost per call: reasoning, remaining output and input (calculation) (Remaining output (visible-answer estimate))",0.003025,"usd","$0.0030",16,"\u0001","\u0001","GPT-6.1 Sol · effort medium","\u0001",true,"\u0001","none"],["thinking-token-bill","thinking-bill-cost-per-call","Remaining output (visible-answer estimate)","GPT-6.1 Sol (high) · Codex CLI","thinking-bill-cost-per-call/Remaining output (visible-answer estimate)","List-price cost per call: reasoning, remaining output and input (calculation) (Remaining output (visible-answer estimate))",0.002905,"usd","$0.0029",16,"\u0001","\u0001","GPT-6.1 Sol · effort high","\u0001",true,"\u0001","none"],["thinking-token-bill","thinking-bill-cost-per-call","Input (prompt, cache priced)","GPT-6.1 Sol (medium) · Codex CLI","thinking-bill-cost-per-call/Input (prompt, cache priced)","List-price cost per call: reasoning, remaining output and input (calculation) (Input (prompt, cache priced))",0.020339,"usd","$0.020",16,"\u0001","\u0001","GPT-6.1 Sol · effort medium","\u0001",true,"\u0001","none"],["thinking-token-bill","thinking-bill-cost-per-call","Input (prompt, cache priced)","GPT-6.1 Sol (high) · Codex CLI","thinking-bill-cost-per-call/Input (prompt, cache priced)","List-price cost per call: reasoning, remaining output and input (calculation) (Input (prompt, cache priced))",0.008117,"usd","$0.0081",16,"\u0001","\u0001","GPT-6.1 Sol · effort high","\u0001",true,"\u0001","none"],["thinking-token-bill","thinking-bill-by-effort","Reasoning cost per strict pass","GPT-6.1 Sol (low) · Codex CLI","thinking-bill-by-effort/Reasoning cost per strict pass","Reasoning cost per strict pass by effort, with the total (calculation) (Reasoning cost per strict pass)",0.001223,"usd","$0.0012",16,"\u0001","\u0001","GPT-6.1 Sol · effort low","\u0001",true,"\u0001","none"],["thinking-token-bill","thinking-bill-by-effort","Reasoning cost per strict pass","GPT-6.1 Sol (medium) · Codex CLI","thinking-bill-by-effort/Reasoning cost per strict pass","Reasoning cost per strict pass by effort, with the total (calculation) (Reasoning cost per strict pass)",0.002273,"usd","$0.0023",16,"\u0001","\u0001","GPT-6.1 Sol · effort medium","\u0001",true,"\u0001","none"],["thinking-token-bill","thinking-bill-by-effort","Reasoning cost per strict pass","GPT-6.1 Sol (high) · Codex CLI","thinking-bill-by-effort/Reasoning cost per strict pass","Reasoning cost per strict pass by effort, with the total (calculation) (Reasoning cost per strict pass)",0.004114,"usd","$0.0041",16,"\u0001","\u0001","GPT-6.1 Sol · effort high","\u0001",true,"\u0001","none"],["thinking-token-bill","thinking-bill-by-effort","Total cost per strict pass","GPT-6.1 Sol (low) · Codex CLI","thinking-bill-by-effort/Total cost per strict pass","Reasoning cost per strict pass by effort, with the total (calculation) (Total cost per strict pass)",0.012837,"usd","$0.013",16,"\u0001","\u0001","GPT-6.1 Sol · effort low","\u0001",true,"\u0001","none"],["thinking-token-bill","thinking-bill-by-effort","Total cost per strict pass","GPT-6.1 Sol (medium) · Codex CLI","thinking-bill-by-effort/Total cost per strict pass","Reasoning cost per strict pass by effort, with the total (calculation) (Total cost per strict pass)",0.025637,"usd","$0.026",16,"\u0001","\u0001","GPT-6.1 Sol · effort medium","\u0001",true,"\u0001","none"],["thinking-token-bill","thinking-bill-by-effort","Total cost per strict pass","GPT-6.1 Sol (high) · Codex CLI","thinking-bill-by-effort/Total cost per strict pass","Reasoning cost per strict pass by effort, with the total (calculation) (Total cost per strict pass)",0.015137,"usd","$0.015",16,"\u0001","\u0001","GPT-6.1 Sol · effort high","\u0001",true,"\u0001","none"],["thinking-token-bill","thinking-bill-short-vs-hard","Eight hard tasks","GPT-6.1 Sol (medium) · Codex CLI","thinking-bill-short-vs-hard/Eight hard tasks","Reasoning share on short tasks vs hard tasks (calculation) (Eight hard tasks)",46.33,"percent","46.3%",16,"\u0001","minmax","GPT-6.1 Sol · effort medium",[11.42,86.85],true,"\u0001","none"],["thinking-token-bill","thinking-bill-short-vs-hard","Eight hard tasks","GPT-6.1 Sol (high) · Codex CLI","thinking-bill-short-vs-hard/Eight hard tasks","Reasoning share on short tasks vs hard tasks (calculation) (Eight hard tasks)",57.01,"percent","57%",16,"\u0001","minmax","GPT-6.1 Sol · effort high",[29.19,90.8],true,"\u0001","none"],["thinking-token-bill","thinking-bill-short-vs-hard","Five short tasks","GPT-6.1 Sol (medium) · Codex CLI","thinking-bill-short-vs-hard/Five short tasks","Reasoning share on short tasks vs hard tasks (calculation) (Five short tasks)",41.05,"percent","41%",15,"\u0001","minmax","GPT-6.1 Sol · effort medium",[0,71.43],true,"\u0001","none"],["thinking-token-bill","thinking-bill-short-vs-hard","Five short tasks","GPT-6.1 Sol (high) · Codex CLI","thinking-bill-short-vs-hard/Five short tasks","Reasoning share on short tasks vs hard tasks (calculation) (Five short tasks)",58.06,"percent","58.1%",15,"\u0001","minmax","GPT-6.1 Sol · effort high",[0,75.76],true,"\u0001","none"],["llm-speed-anatomy","speed-anatomy-first-text","Time to first text","GPT-6.1 Sol (low) · Codex CLI","speed-anatomy-first-text","Time to first text: a 250-line answer, six models",3.52,"seconds","3.52 s",4,"\u0001","minmax","GPT-6.1 Sol · effort low",[2.75,4.42],"\u0001","\u0001","\u0001"],["llm-speed-anatomy","speed-anatomy-first-text","Time to first text","GPT-6 Luna (low) · Codex CLI","speed-anatomy-first-text","Time to first text: a 250-line answer, six models",3.3,"seconds","3.30 s",4,"\u0001","minmax","GPT-6 Luna · effort low",[3.19,3.47],"\u0001","\u0001","\u0001"],["llm-speed-anatomy","speed-anatomy-output-speed","Visible tokens per second","GPT-6.1 Sol (low) · Codex CLI","speed-anatomy-output-speed","Output speed after the first text: visible tokens per second (calculation)",79.6,"tokens","80",4,"\u0001","minmax","GPT-6.1 Sol · effort low",[71.6,80.5],true,"\u0001","\u0001"],["llm-speed-anatomy","speed-anatomy-output-speed","Visible tokens per second","GPT-6 Luna (low) · Codex CLI","speed-anatomy-output-speed","Output speed after the first text: visible tokens per second (calculation)",129.1,"tokens","129",4,"\u0001","minmax","GPT-6 Luna · effort low",[55.5,259.1],true,"\u0001","\u0001"],["llm-speed-anatomy","speed-anatomy-chars-per-second","Characters per second","GPT-6.1 Sol (low) · Codex CLI","speed-anatomy-chars-per-second","Output speed in characters per second after the first text (calculation)",323,"count","323",4,"\u0001","minmax","GPT-6.1 Sol · effort low",[291,327],true,"\u0001","\u0001"],["llm-speed-anatomy","speed-anatomy-chars-per-second","Characters per second","GPT-6 Luna (low) · Codex CLI","speed-anatomy-chars-per-second","Output speed in characters per second after the first text (calculation)",524,"count","524",4,"\u0001","minmax","GPT-6 Luna · effort low",[225,1052],true,"\u0001","\u0001"],["llm-speed-anatomy","speed-anatomy-prompt-size","GPT-6.1 Sol (low) · Codex CLI","1k","speed-anatomy-prompt-size/1k","Time to first text as the prompt grows: 1k",3.36,"seconds","3.36 s",3,"\u0001","minmax","GPT-6.1 Sol · effort low",[3.36,4.75],true,"\u0001","\u0001"],["llm-speed-anatomy","speed-anatomy-prompt-size","GPT-6.1 Sol (low) · Codex CLI","16k","speed-anatomy-prompt-size/16k","Time to first text as the prompt grows: 16k",4.02,"seconds","4.02 s",3,"\u0001","minmax","GPT-6.1 Sol · effort low",[3.3,4.28],true,"\u0001","\u0001"],["llm-speed-anatomy","speed-anatomy-prompt-size","GPT-6.1 Sol (low) · Codex CLI","64k","speed-anatomy-prompt-size/64k","Time to first text as the prompt grows: 64k",3.93,"seconds","3.93 s",3,"\u0001","minmax","GPT-6.1 Sol · effort low",[3.42,4.38],true,"\u0001","\u0001"],["llm-speed-anatomy","speed-anatomy-total-by-size","1k prompt","GPT-6.1 Sol (low) · Codex CLI","speed-anatomy-total-by-size/1k prompt","Total time per call by prompt size (1k prompt)",3.43,"seconds","3.43 s",3,"\u0001","minmax","GPT-6.1 Sol · effort low",[3.43,4.92],"\u0001","\u0001","\u0001"],["llm-speed-anatomy","speed-anatomy-total-by-size","16k prompt","GPT-6.1 Sol (low) · Codex CLI","speed-anatomy-total-by-size/16k prompt","Total time per call by prompt size (16k prompt)",4.14,"seconds","4.14 s",3,"\u0001","minmax","GPT-6.1 Sol · effort low",[3.96,4.68],"\u0001","\u0001","\u0001"],["llm-speed-anatomy","speed-anatomy-total-by-size","64k prompt","GPT-6.1 Sol (low) · Codex CLI","speed-anatomy-total-by-size/64k prompt","Total time per call by prompt size (64k prompt)",3.96,"seconds","3.96 s",3,"\u0001","minmax","GPT-6.1 Sol · effort low",[3.47,4.44],"\u0001","\u0001","\u0001"],["llm-speed-anatomy","speed-anatomy-lookup-correct","Exact answer","GPT-6.1 Sol (low) · Codex CLI","speed-anatomy-lookup-correct","Exact lookup answers at the 1k, 16k and 64k prompt-size targets",1,"rate","100% (9/9)",9,[0.7009,1],"ci95","GPT-6.1 Sol · effort low","\u0001","\u0001","\u0001","higher"],["harder-tasks-head-to-head","harder-h2h-pass-rate","Strict pass","GPT-6.1 Sol (medium) · Codex CLI","harder-h2h-pass-rate/Strict pass","Pass rate on 4 harder tasks (Strict pass)",0.6875,"rate","69% (11/16)",16,[0.444,0.8584],"ci95","GPT-6.1 Sol · effort medium","\u0001","\u0001","\u0001","higher"],["harder-tasks-head-to-head","harder-h2h-pass-rate","Lenient (format misses counted)","GPT-6.1 Sol (medium) · Codex CLI","harder-h2h-pass-rate/Lenient (format misses counted)","Pass rate on 4 harder tasks (Lenient (format misses counted))",0.6875,"rate","69% (11/16)",16,[0.444,0.8584],"ci95","GPT-6.1 Sol · effort medium","\u0001","\u0001","\u0001","higher"],["harder-tasks-head-to-head","harder-h2h-tool-attempts","Tool attempt","GPT-6.1 Sol (medium) · Codex CLI","harder-h2h-tool-attempts","Calls that tried a tool although tools were off",0,"rate","0% (0/16)",16,[0,0.1936],"ci95","GPT-6.1 Sol · effort medium","\u0001","\u0001","\u0001","none"],["harder-tasks-head-to-head","harder-h2h-pass-by-task","GPT-6.1 Sol (medium) · Codex CLI","10x10 nonogram","harder-h2h-pass-by-task/10x10 nonogram","Strict pass rate by task: 10x10 nonogram",0.75,"rate","75% (3/4)",4,[0.3006,0.9544],"ci95","GPT-6.1 Sol · effort medium","\u0001","\u0001","\u0001","higher"],["harder-tasks-head-to-head","harder-h2h-pass-by-task","GPT-6.1 Sol (medium) · Codex CLI","Sudoku, 22 givens","harder-h2h-pass-by-task/Sudoku, 22 givens","Strict pass rate by task: Sudoku, 22 givens",0.25,"rate","25% (1/4)",4,[0.0456,0.6994],"ci95","GPT-6.1 Sol · effort medium","\u0001","\u0001","\u0001","higher"],["harder-tasks-head-to-head","harder-h2h-pass-by-task","GPT-6.1 Sol (medium) · Codex CLI","6x6 Skyscrapers","harder-h2h-pass-by-task/6x6 Skyscrapers","Strict pass rate by task: 6x6 Skyscrapers",1,"rate","100% (4/4)",4,[0.5101,1],"ci95","GPT-6.1 Sol · effort medium","\u0001","\u0001","\u0001","higher"],["harder-tasks-head-to-head","harder-h2h-pass-by-task","GPT-6.1 Sol (medium) · Codex CLI","Seeded shuffle output","harder-h2h-pass-by-task/Seeded shuffle output","Strict pass rate by task: Seeded shuffle output",0.75,"rate","75% (3/4)",4,[0.3006,0.9544],"ci95","GPT-6.1 Sol · effort medium","\u0001","\u0001","\u0001","higher"],["harder-tasks-head-to-head","harder-h2h-total-latency","Total time per call","GPT-6.1 Sol (medium) · Codex CLI","harder-h2h-total-latency","Total time per call on harder tasks",120.24,"seconds","120.2 s",13,"\u0001","minmax","GPT-6.1 Sol · effort medium",[46.24,273.46],"\u0001","\u0001","\u0001"],["harder-tasks-head-to-head","harder-h2h-output-tokens","Output tokens","GPT-6.1 Sol (medium) · Codex CLI","harder-h2h-output-tokens/Output tokens","Output tokens per call on harder tasks (Output tokens)",4994,"tokens","4,994",13,"\u0001","minmax","GPT-6.1 Sol · effort medium",[2099,13413],"\u0001","\u0001","none"],["harder-tasks-head-to-head","harder-h2h-cost-per-pass","Cost per strict pass","GPT-6.1 Sol (medium) · Codex CLI","harder-h2h-cost-per-pass","List-price cost per strict pass on harder tasks (calculation)",0.08293,"usd","$0.083",16,"\u0001","\u0001","GPT-6.1 Sol · effort medium","\u0001",true,"\u0001","\u0001"]]}],["jev-1-13","Jev 1.13","TypeSafe","router","A small routing model from TypeSafe that answers typed routing decisions over an API. Priced on input tokens only. Measured here as a router (accuracy, live time per call and cost per 1,000 decisions) and in the System One arena.",["Jev 1.13"],{"$k":["studySlug","chartId","series","point","metric","label","value","unit","display","n","ci","spanKind","context","range","polarity","calculation","statId"],"$r":[["system-one-arena","arena-accuracy","Accuracy","Jev 1.13","arena-accuracy","Who decides right? Accuracy on 1,000+ checkable decisions",0.7681,"rate","77% (805/1048)",1048,[0.7416,0.7927],"ci95","","\u0001","\u0001","\u0001","\u0001"],["system-one-arena","arena-by-suite","Jev 1.13","Games","arena-by-suite/Games","Where each model is strong: Games",0.4219,"rate","42% (81/192)",192,[0.3542,0.4926],"ci95","","\u0001","\u0001","\u0001","\u0001"],["system-one-arena","arena-by-suite","Jev 1.13","Logic and thought experiments","arena-by-suite/Logic and thought experiments","Where each model is strong: Logic and thought experiments",0.7436,"rate","74% (116/156)",156,[0.6698,0.8057],"ci95","","\u0001","\u0001","\u0001","\u0001"],["system-one-arena","arena-by-suite","Jev 1.13","Policy cases, 20 industries","arena-by-suite/Policy cases, 20 industries","Where each model is strong: Policy cases, 20 industries",0.8139,"rate","81% (223/274)",274,[0.7636,0.8555],"ci95","","\u0001","\u0001","\u0001","\u0001"],["system-one-arena","arena-by-suite","Jev 1.13","Usability intents","arena-by-suite/Usability intents","Where each model is strong: Usability intents",0.9336,"rate","93% (211/226)",226,[0.8934,0.9594],"ci95","","\u0001","\u0001","\u0001","\u0001"],["system-one-arena","arena-by-suite","Jev 1.13","Stress tests","arena-by-suite/Stress tests","Where each model is strong: Stress tests",0.87,"rate","87% (174/200)",200,[0.8163,0.9097],"ci95","","\u0001","\u0001","\u0001","\u0001"],["system-one-arena","arena-latency-same-gpu","Hosted: network included","Jev 1.13","arena-latency-same-gpu/Hosted: network included","Speed on one GPU: third-party numbers (Hosted: network included)",524.1,"ms","524 ms","\u0001","\u0001","p50-p95","",[524.1,536],"lower","\u0001","\u0001"],["system-one-arena","arena-latency-gateway","Median latency, one gateway","Jev 1.13 (TypeSafe)","arena-latency-gateway","Speed through one gateway: OpenRouter's own numbers",0.17,"seconds","0.17 s","\u0001","\u0001","\u0001","","\u0001","lower","\u0001","\u0001"],["system-one-arena","arena-cost-same-provider","USD per 1,000 decisions","Jev 1.13 ($0.042/M, 825 tokens)","arena-cost-same-provider","Price per 1,000 decisions at one provider's list prices",0.0347,"usd","$0.035","\u0001","\u0001","\u0001","$0.042/M · 825 tokens","\u0001","lower",true,"\u0001"],["system-one-arena","arena-robust","First presentation","Jev 1.13","arena-robust/First presentation","Same question, different presentation (First presentation)",0.7681,"rate","77% (805/1048)",1048,[0.7416,0.7927],"ci95","","\u0001","\u0001","\u0001","\u0001"],["system-one-arena","arena-robust","All three presentations","Jev 1.13","arena-robust/All three presentations","Same question, different presentation (All three presentations)",0.729,"rate","73% (764/1048)",1048,[0.7013,0.755],"ci95","","\u0001","\u0001","\u0001","\u0001"],["system-one-arena","arena-flips","Options shuffled","Jev 1.13","arena-flips/Options shuffled","Decisions that changed when only the presentation changed (Options shuffled)",0.077,"rate","8% (74/961)",961,[0.0618,0.0956],"ci95","","\u0001","lower","\u0001","\u0001"],["system-one-arena","arena-flips","Options renamed a, b, c","Jev 1.13","arena-flips/Options renamed a, b, c","Decisions that changed when only the presentation changed (Options renamed a, b, c)",0.0687,"rate","7% (66/961)",961,[0.0543,0.0864],"ci95","","\u0001","lower","\u0001","\u0001"],["system-one-arena","arena-escape","Escaped when it should (higher is better)","Jev 1.13","arena-escape/Escaped when it should (higher is better)","Knowing when to say \"none of these\" (Escaped when it should (higher is better))",0.8571,"rate","86% (126/147)",147,[0.7915,0.9046],"ci95","","\u0001","none","\u0001","\u0001"],["system-one-arena","arena-escape","Escaped when it should not (lower is better)","Jev 1.13","arena-escape/Escaped when it should not (lower is better)","Knowing when to say \"none of these\" (Escaped when it should not (lower is better))",0.055,"rate","6% (34/618)",618,[0.0396,0.0759],"ci95","","\u0001","none","\u0001","\u0001"],["system-one-arena","arena-injection","Right despite the injected text","Jev 1.13","arena-injection","Prompt injection: does text in the state hijack the decision?",0.95,"rate","95% (38/40)",40,[0.835,0.9862],"ci95","","\u0001","\u0001","\u0001","\u0001"],["system-one-arena","arena-by-length","Up to 512 tokens","Jev 1.13","arena-by-length/Up to 512 tokens","Short inputs vs long inputs (Up to 512 tokens)",0.7867,"rate","79% (177/225)",225,[0.7286,0.8351],"ci95","","\u0001","\u0001","\u0001","\u0001"],["system-one-arena","arena-by-length","More than 512 tokens","Jev 1.13","arena-by-length/More than 512 tokens","Short inputs vs long inputs (More than 512 tokens)",0.7631,"rate","76% (628/823)",823,[0.7328,0.7908],"ci95","","\u0001","\u0001","\u0001","\u0001"],["system-one-arena","arena-confident-wrong","Confident wrong answers","Jev 1.13","arena-confident-wrong","Wrong and sure of it",0.1111,"rate","11% (27/243)",243,[0.0775,0.1568],"ci95","","\u0001","lower","\u0001","\u0001"],["system-one-arena","arena-elo","Elo","Jev 1.13","arena-elo","Tournament rating across every game",1040,"score","1040.00",336,"\u0001","minmax","",[1002,1082],"\u0001","\u0001","\u0001"],["system-one-arena","arena-wdl","Wins","Jev 1.13","arena-wdl/Wins","Wins, draws and losses in the round robin (Wins)",196,"count","196",336,"\u0001","\u0001","","\u0001","\u0001","\u0001","\u0001"],["system-one-arena","arena-wdl","Draws","Jev 1.13","arena-wdl/Draws","Wins, draws and losses in the round robin (Draws)",6,"count","6",336,"\u0001","\u0001","","\u0001","\u0001","\u0001","\u0001"],["system-one-arena","arena-wdl","Losses","Jev 1.13","arena-wdl/Losses","Wins, draws and losses in the round robin (Losses)",134,"count","134",336,"\u0001","\u0001","","\u0001","\u0001","\u0001","\u0001"],["system-one-arena","arena-perfect-moves","Jev 1.13","Tic-tac-toe","arena-perfect-moves/Tic-tac-toe","How often a model found the perfect move: Tic-tac-toe",0.4653,"rate","47% (67/144)",144,[0.3858,0.5466],"ci95","","\u0001","\u0001","\u0001","\u0001"],["system-one-arena","arena-perfect-moves","Jev 1.13","Connect Four","arena-perfect-moves/Connect Four","How often a model found the perfect move: Connect Four",0.3789,"rate","38% (133/351)",351,[0.3297,0.4307],"ci95","","\u0001","\u0001","\u0001","\u0001"],["system-one-arena","arena-perfect-moves","Jev 1.13","Nim","arena-perfect-moves/Nim","How often a model found the perfect move: Nim",0.3916,"rate","39% (65/166)",166,[0.3206,0.4675],"ci95","","\u0001","\u0001","\u0001","\u0001"],["system-one-arena","arena-perfect-moves","Jev 1.13","Dots and Boxes","arena-perfect-moves/Dots and Boxes","How often a model found the perfect move: Dots and Boxes",0.5479,"rate","55% (423/772)",772,[0.5127,0.5827],"ci95","","\u0001","\u0001","\u0001","\u0001"],["system-one-arena","arena-pong","Pong score","Jev 1.13","arena-pong","Pong as deployed: who won",0.8125,"rate","81% (13/16)",16,[0.5699,0.9341],"ci95","","\u0001","\u0001","\u0001","\u0001"],["system-one-arena","arena-pong-quality","Right zone","Jev 1.13","arena-pong-quality","Pong decision quality, ignoring time",1,"rate","100% (523/523)",523,[0.9927,1],"ci95","","\u0001","\u0001","\u0001","\u0001"],["system-one-arena","\u0001","\u0001","\u0001","stat:arena-jev-latency","Jev 1.13 hosted API: median call time from Houston, network included (not comparable with a model on another machine)",137,"ms","137 ms",120,"\u0001","\u0001","","\u0001","\u0001","\u0001","arena-jev-latency"],["routing-jev-vs-llm","routing-exact-decisions","Exact rate","Jev 1.13 (TypeSafe)","routing-exact-decisions","Typed routing decisions answered exactly right",0.8984,"rate","90%",82,[0.8191,0.9497],"ci95","typed routing decisions · TypeSafe API","\u0001","\u0001","\u0001","\u0001"],["routing-jev-vs-llm","routing-key-accuracy","Key accuracy","Jev 1.13 (TypeSafe)","routing-key-accuracy","Per-question accuracy",0.9485,"rate","95% (184/194)",194,[0.9077,0.9718],"ci95","typed routing decisions · TypeSafe API","\u0001","\u0001","\u0001","\u0001"],["routing-jev-vs-llm","routing-exact-by-decision","Jev 1.13 (TypeSafe)","Failure class","routing-exact-by-decision/Failure class","Exact rate by decision type: Failure class",1,"rate","100% (18/18)",18,[0.8241,1],"ci95","typed routing decisions · TypeSafe API","\u0001","\u0001","\u0001","\u0001"],["routing-jev-vs-llm","routing-exact-by-decision","Jev 1.13 (TypeSafe)","Message intent","routing-exact-by-decision/Message intent","Exact rate by decision type: Message intent",1,"rate","100% (20/20)",20,[0.8389,1],"ci95","typed routing decisions · TypeSafe API","\u0001","\u0001","\u0001","\u0001"],["routing-jev-vs-llm","routing-exact-by-decision","Jev 1.13 (TypeSafe)","Is it a rule?","routing-exact-by-decision/Is it a rule?","Exact rate by decision type: Is it a rule?",1,"rate","100% (12/12)",12,[0.7575,1],"ci95","typed routing decisions · TypeSafe API","\u0001","\u0001","\u0001","\u0001"],["routing-jev-vs-llm","routing-exact-by-decision","Jev 1.13 (TypeSafe)","Context shape","routing-exact-by-decision/Context shape","Exact rate by decision type: Context shape",0.7396,"rate","74%",32,[0.5789,0.8675],"ci95","typed routing decisions · TypeSafe API","\u0001","\u0001","\u0001","\u0001"],["routing-jev-vs-llm","routing-cost-per-1000","Cost","Jev 1.13 (TypeSafe)","routing-cost-per-1000","Cost per 1,000 routing decisions",0.0337,"usd","$0.034",246,"\u0001","\u0001","typed routing decisions · TypeSafe API","\u0001","\u0001",true,"\u0001"],["routing-jev-vs-llm","routing-decision-latency","Wall time (direct API call)","Jev 1.13 (TypeSafe)","routing-decision-latency/Wall time (direct API call)","Time per routing decision (Wall time (direct API call))",136.5,"ms","137 ms",246,"\u0001","p50-p95","typed routing decisions · TypeSafe API",[136.5,195.7],"\u0001","\u0001","\u0001"],["routing-overhead","router-overhead-decision-latency","Decision time","Jev 1.13 (TypeSafe)","router-overhead-decision-latency","Time to make one routing decision",136.5,"ms","137 ms",246,"\u0001","p50-p95","routing overhead per decision · TypeSafe API",[136.5,195.7],"\u0001","\u0001","\u0001"],["routing-overhead","router-overhead-completed","Completed","Jev 1.13 (TypeSafe)","router-overhead-completed","Routing calls that returned a decision",1,"rate","100% (246/246)",246,[0.9846,1],"ci95","routing overhead per decision · TypeSafe API","\u0001","\u0001","\u0001","\u0001"],["routing-overhead","router-overhead-cost-reported","Cost per 1,000 decisions","Jev 1.13 (TypeSafe)","router-overhead-cost-reported","Cost per 1,000 routing decisions: no model call vs provider-reported",0.0337,"usd","$0.034",82,"\u0001","\u0001","routing overhead per decision · TypeSafe API","\u0001","\u0001","\u0001","\u0001"],["routing-overhead","router-overhead-cost-list-price","Cost per 1,000 decisions (list price)","Jev 1.13 (TypeSafe)","router-overhead-cost-list-price","Cost per 1,000 routing decisions for the model routers (calculation)",0.0337,"usd","$0.034",246,"\u0001","\u0001","calculation: Agent’s recorded tokens at this model’s list price · routing overhead per decision · TypeSafe API","\u0001","\u0001",true,"\u0001"],["routing-overhead","router-overhead-cost-per-1000-tasks","Every model call routed (49.5 per task)","Jev 1.13 (TypeSafe)","router-overhead-cost-per-1000-tasks/Every model call routed (49.5 per task)","Added routing cost per 1,000 tasks (calculation) (Every model call routed (49.5 per task))",1.67,"usd","$1.67","\u0001","\u0001","\u0001","calculation per 1,000 tasks from recorded decision counts · TypeSafe API","\u0001","\u0001",true,"\u0001"],["routing-overhead","router-overhead-cost-per-1000-tasks","Only System One decisions (7 per task)","Jev 1.13 (TypeSafe)","router-overhead-cost-per-1000-tasks/Only System One decisions (7 per task)","Added routing cost per 1,000 tasks (calculation) (Only System One decisions (7 per task))",0.24,"usd","$0.24","\u0001","\u0001","\u0001","calculation per 1,000 tasks from recorded decision counts · TypeSafe API","\u0001","\u0001",true,"\u0001"],["routing-overhead","router-overhead-delay-per-task","Every model call routed (49.5 per task)","Jev 1.13 (TypeSafe)","router-overhead-delay-per-task/Every model call routed (49.5 per task)","Added routing delay per task (calculation) (Every model call routed (49.5 per task))",6.7568,"seconds","6.76 s","\u0001","\u0001","\u0001","calculation per task from recorded decision counts, decisions in line · TypeSafe API","\u0001","\u0001",true,"\u0001"],["routing-overhead","router-overhead-delay-per-task","Only System One decisions (7 per task)","Jev 1.13 (TypeSafe)","router-overhead-delay-per-task/Only System One decisions (7 per task)","Added routing delay per task (calculation) (Only System One decisions (7 per task))",0.9555,"seconds","0.96 s","\u0001","\u0001","\u0001","calculation per task from recorded decision counts, decisions in line · TypeSafe API","\u0001","\u0001",true,"\u0001"],["cost-thought-experiments","repriced-cost-per-resolved","Repriced cost per resolved instance","Jev 1.13 (router)","repriced-cost-per-resolved","Thought experiment: the same tokens at other list prices",0.042,"usd","$0.042","\u0001","\u0001","\u0001","router · calculation: Agent’s recorded tokens at this model’s list price","\u0001","\u0001",true,"\u0001"],["routing-holdout","routing-holdout-exact","Exact rate","Jev 1.13 (TypeSafe)","routing-holdout-exact","Unseen routing decisions answered exactly right",0.8214,"rate","82% (46/56)",56,[0.7016,0.9],"ci95","","\u0001","higher","\u0001","\u0001"],["routing-holdout","routing-holdout-key-accuracy","Key accuracy","Jev 1.13 (TypeSafe)","routing-holdout-key-accuracy","Per-question accuracy on unseen decisions",0.904,"rate","90% (113/125)",125,[0.8397,0.9442],"ci95","","\u0001","higher","\u0001","\u0001"],["routing-holdout","routing-holdout-by-purpose","Jev 1.13 (TypeSafe)","Failure class","routing-holdout-by-purpose/Failure class","Exact rate on unseen decisions, by decision type: Failure class",0.9286,"rate","93% (13/14)",14,[0.6853,0.9873],"ci95","","\u0001","higher","\u0001","\u0001"],["routing-holdout","routing-holdout-by-purpose","Jev 1.13 (TypeSafe)","Message intent","routing-holdout-by-purpose/Message intent","Exact rate on unseen decisions, by decision type: Message intent",0.8571,"rate","86% (12/14)",14,[0.6006,0.9599],"ci95","","\u0001","higher","\u0001","\u0001"],["routing-holdout","routing-holdout-by-purpose","Jev 1.13 (TypeSafe)","Is it a rule?","routing-holdout-by-purpose/Is it a rule?","Exact rate on unseen decisions, by decision type: Is it a rule?",0.9286,"rate","93% (13/14)",14,[0.6853,0.9873],"ci95","","\u0001","higher","\u0001","\u0001"],["routing-holdout","routing-holdout-by-purpose","Jev 1.13 (TypeSafe)","Context shape","routing-holdout-by-purpose/Context shape","Exact rate on unseen decisions, by decision type: Context shape",0.5714,"rate","57% (8/14)",14,[0.3259,0.7862],"ci95","","\u0001","higher","\u0001","\u0001"],["routing-holdout","routing-holdout-tuned-vs-unseen","Tuned set (routing-jev-vs-llm)","Jev 1.13 (TypeSafe)","routing-holdout-tuned-vs-unseen/Tuned set (routing-jev-vs-llm)","Tuned case set vs unseen holdout: exact rate per router (Tuned set (routing-jev-vs-llm))",0.9024,"rate","90% (74/82)",82,[0.8191,0.9497],"ci95","","\u0001","higher","\u0001","\u0001"],["routing-holdout","routing-holdout-tuned-vs-unseen","Unseen holdout","Jev 1.13 (TypeSafe)","routing-holdout-tuned-vs-unseen/Unseen holdout","Tuned case set vs unseen holdout: exact rate per router (Unseen holdout)",0.8214,"rate","82% (46/56)",56,[0.7016,0.9],"ci95","","\u0001","higher","\u0001","\u0001"],["routing-holdout","routing-holdout-latency","Wall time","Jev 1.13 (TypeSafe)","routing-holdout-latency/Wall time","Time per routing decision, by route (Wall time)",0.139,"seconds","0.14 s",168,"\u0001","p50-p95","",[0.139,0.192],"\u0001","\u0001","\u0001"],["routing-holdout","routing-holdout-cost-per-1000","Cost","Jev 1.13 (TypeSafe)","routing-holdout-cost-per-1000","Cost per 1,000 unseen routing decisions",0.03065,"usd","$0.031",168,"\u0001","\u0001","","\u0001","\u0001",true,"\u0001"],["routing-holdout","\u0001","\u0001","\u0001","stat:holdout-jev-stability","Jev 1.13 (TypeSafe): same answers on every key in 3 repetitions",0.9464,"rate","95% (53/56)",56,[0.8539,0.9816],"ci95","TypeSafe","\u0001","\u0001","\u0001","holdout-jev-stability"],["routing-holdout","\u0001","\u0001","\u0001","stat:holdout-gap-jev","Jev 1.13 (TypeSafe): holdout minus tuned-set exact rate",-0.081,"rate","−8.1 points",56,"\u0001","\u0001","TypeSafe","\u0001","\u0001",true,"holdout-gap-jev"]]}],["agent-harness","Agent","Agent","harness","The Agent coding pipeline: onboarding, research, plan, act, verify and review, on Claude Sonnet 5.5 through a Claude subscription.",["Agent"],{"$k":["studySlug","chartId","series","point","metric","label","value","unit","display","n","ci","spanKind","context","statId","calculation"],"$r":[["swe-bench-verified","swebench-same-instance-leaderboard","Resolved rate","Agent (Sonnet 5.5, full pipeline)","swebench-same-instance-leaderboard","Resolved rate on the same 33 SWE-bench Verified instances",0.7576,"rate","76% (25/33)",33,[0.5898,0.8717],"ci95","full pipeline on Claude Sonnet 5.5","\u0001","\u0001"],["swe-bench-verified","swebench-by-difficulty-band","Agent","No panel model solved it","swebench-by-difficulty-band/No panel model solved it","Resolved rate by difficulty band: No panel model solved it",0.25,"rate","25% (1/4)",4,[0.0456,0.6994],"ci95","full pipeline on Claude Sonnet 5.5","\u0001","\u0001"],["swe-bench-verified","swebench-by-difficulty-band","Agent","Under half solved it","swebench-by-difficulty-band/Under half solved it","Resolved rate by difficulty band: Under half solved it",0.75,"rate","75% (3/4)",4,[0.3006,0.9544],"ci95","full pipeline on Claude Sonnet 5.5","\u0001","\u0001"],["swe-bench-verified","swebench-by-difficulty-band","Agent","Half or more solved it","swebench-by-difficulty-band/Half or more solved it","Resolved rate by difficulty band: Half or more solved it",0.8182,"rate","82% (9/11)",11,[0.523,0.9486],"ci95","full pipeline on Claude Sonnet 5.5","\u0001","\u0001"],["swe-bench-verified","swebench-by-difficulty-band","Agent","Every panel model solved it","swebench-by-difficulty-band/Every panel model solved it","Resolved rate by difficulty band: Every panel model solved it",0.8571,"rate","86% (12/14)",14,[0.6006,0.9599],"ci95","full pipeline on Claude Sonnet 5.5","\u0001","\u0001"],["swe-bench-verified","swebench-model-calls","Mean calls","Agent","swebench-model-calls","Model calls per instance",49.5,"calls","49.5",33,"\u0001","\u0001","full pipeline on Claude Sonnet 5.5","\u0001","\u0001"],["swe-bench-verified","swebench-views","Agent","Campaign 1: 25-instance sample","swebench-views/Campaign 1: 25-instance sample","Every way to slice the run, with intervals: Campaign 1: 25-instance sample",0.72,"rate","72% (18/25)",25,[0.5242,0.8572],"ci95","full pipeline on Claude Sonnet 5.5","\u0001","\u0001"],["swe-bench-verified","swebench-views","Agent","Campaign 2: 8 compiled-extension instances","swebench-views/Campaign 2: 8 compiled-extension instances","Every way to slice the run, with intervals: Campaign 2: 8 compiled-extension instances",0.875,"rate","88% (7/8)",8,[0.5291,0.9776],"ci95","full pipeline on Claude Sonnet 5.5","\u0001","\u0001"],["swe-bench-verified","swebench-views","Agent","Original seed draw of 25","swebench-views/Original seed draw of 25","Every way to slice the run, with intervals: Original seed draw of 25",0.76,"rate","76% (19/25)",25,[0.5657,0.885],"ci95","full pipeline on Claude Sonnet 5.5","\u0001","\u0001"],["swe-bench-verified","swebench-views","Agent","All 33 attempted","swebench-views/All 33 attempted","Every way to slice the run, with intervals: All 33 attempted",0.7576,"rate","76% (25/33)",33,[0.5898,0.8717],"ci95","full pipeline on Claude Sonnet 5.5","\u0001","\u0001"],["swe-bench-verified","\u0001","\u0001","\u0001","stat:cost-per-attempt","Agent model cost per attempt (notional)",2.81,"usd","$2.81",33,"\u0001","\u0001","full pipeline on Claude Sonnet 5.5","cost-per-attempt",true],["swe-bench-verified","\u0001","\u0001","\u0001","stat:cost-per-resolved","Agent model cost per resolved instance (notional)",3.71,"usd","$3.71",25,"\u0001","\u0001","full pipeline on Claude Sonnet 5.5","cost-per-resolved",true],["swe-bench-verified","\u0001","\u0001","\u0001","stat:median-minutes","Median worker time per attempt",9.6,"minutes","9.6 min",33,"\u0001","\u0001","full pipeline on Claude Sonnet 5.5","median-minutes","\u0001"],["blind-review-head-to-head","\u0001","\u0001","\u0001","stat:ai-preferred-latest","Tasks where the panel preferred the AI change (latest attempt)",0.75,"rate","75% (9/12)",12,[0.4677,0.9111],"ci95","blind panel: Agent change vs merged human change · blind review panel","ai-preferred-latest","\u0001"],["blind-review-head-to-head","\u0001","\u0001","\u0001","stat:ai-preferred-first","Tasks where the panel preferred the AI change (first scored attempt)",0.5,"rate","50% (6/12)",12,[0.2538,0.7462],"ci95","blind panel: Agent change vs merged human change · blind review panel","ai-preferred-first","\u0001"],["blind-review-head-to-head","\u0001","\u0001","\u0001","stat:ai-preferred-all-pairs","All scored pairs where the panel preferred the AI change",0.6,"rate","60% (12/20)",20,[0.3866,0.7812],"ci95","blind panel: Agent change vs merged human change · blind review panel","ai-preferred-all-pairs","\u0001"],["blind-review-head-to-head","\u0001","\u0001","\u0001","stat:ai-preferred-public","Public OSS tasks, latest attempt",0.8,"rate","80% (4/5)",5,[0.3755,0.9638],"ci95","blind panel: Agent change vs merged human change · blind review panel","ai-preferred-public","\u0001"],["blind-review-head-to-head","\u0001","\u0001","\u0001","stat:verdicts-ai","Single critic verdicts that preferred the AI change",0.6894,"rate","69% (91/132)",132,[0.606,0.762],"ci95","blind panel: Agent change vs merged human change · blind review panel","verdicts-ai","\u0001"],["cost-thought-experiments","cost-per-resolved-agent-vs-panel","Cost per resolved instance","Agent (notional)","cost-per-resolved-agent-vs-panel","Recorded cost per resolved instance: Agent vs the public panel",3.706,"usd","$3.71",25,"\u0001","\u0001","full pipeline on Claude Sonnet 5.5 · notional","\u0001",true],["coding-calibration","\u0001","\u0001","\u0001","stat:verified-latest","Verified deliveries, latest build",1,"count","1 of 3",3,"\u0001","\u0001","three real tasks, platform builds compared","verified-latest","\u0001"],["coding-calibration","\u0001","\u0001","\u0001","stat:cost-latest","Notional cost, latest build, all 3 tasks",11.06,"usd","$11.06",3,"\u0001","\u0001","three real tasks, platform builds compared","cost-latest",true],["coding-calibration","\u0001","\u0001","\u0001","stat:refusals-trend","Guardrail refusals, first vs latest slice",19,"count","26 → 19",3,"\u0001","\u0001","three real tasks, platform builds compared","refusals-trend","\u0001"],["coding-calibration","\u0001","\u0001","\u0001","stat:first-run-cost","First calibration run (capped, fastify/session)",4.89,"usd","$4.89, stopped at cap",1,"\u0001","\u0001","three real tasks, platform builds compared","first-run-cost",true]]}],["gpt-6-1-sol-openai-api","GPT-6.1 Sol (OpenAI API)","OpenAI","model","OpenAI’s GPT-6.1 Sol model called directly through the OpenAI API, without a CLI.",["GPT-6.1 Sol · OpenAI API"],{"$k":["studySlug","chartId","series","point","metric","label","value","unit","display","n","range","spanKind","context"],"$r":[["cli-model-latency-tokens","cli-vs-api-exact-reply-latency","Total time","OpenAI API · GPT-6.1 Sol · low","cli-vs-api-exact-reply-latency/Total time","CLI vs API: time for a one-line answer (Total time)",1.02,"seconds","1.02 s",5,[0.96,1.87],"minmax","OpenAI API · effort low · fixed exact reply, 5 runs"],["cli-model-latency-tokens","cli-vs-api-exact-reply-latency","Total time","OpenAI API · GPT-6.1 Sol · high","cli-vs-api-exact-reply-latency/Total time","CLI vs API: time for a one-line answer (Total time)",1.52,"seconds","1.52 s",5,[1.35,2.23],"minmax","OpenAI API · effort high · fixed exact reply, 5 runs"],["cli-model-latency-tokens","cli-vs-api-exact-reply-latency","First useful output","OpenAI API · GPT-6.1 Sol · low","cli-vs-api-exact-reply-latency/First useful output","CLI vs API: time for a one-line answer (First useful output)",0.87,"seconds","0.87 s",5,[0.84,1.74],"minmax","OpenAI API · effort low · fixed exact reply, 5 runs"],["cli-model-latency-tokens","cli-vs-api-exact-reply-latency","First useful output","OpenAI API · GPT-6.1 Sol · high","cli-vs-api-exact-reply-latency/First useful output","CLI vs API: time for a one-line answer (First useful output)",1.34,"seconds","1.34 s",5,[1.26,2.12],"minmax","OpenAI API · effort high · fixed exact reply, 5 runs"],["cli-model-latency-tokens","cli-vs-api-small-coding-latency","Total time","OpenAI API · GPT-6.1 Sol · low","cli-vs-api-small-coding-latency/Total time","CLI vs API: time for a small coding task (Total time)",6,"seconds","6.00 s",3,[5.44,6.2],"minmax","OpenAI API · effort low · small coding task, 3 runs"],["cli-model-latency-tokens","cli-vs-api-small-coding-latency","Total time","OpenAI API · GPT-6.1 Sol · high","cli-vs-api-small-coding-latency/Total time","CLI vs API: time for a small coding task (Total time)",9.56,"seconds","9.56 s",3,[9.44,10.94],"minmax","OpenAI API · effort high · small coding task, 3 runs"],["cli-model-latency-tokens","cli-vs-api-small-coding-latency","First useful output","OpenAI API · GPT-6.1 Sol · low","cli-vs-api-small-coding-latency/First useful output","CLI vs API: time for a small coding task (First useful output)",1.05,"seconds","1.05 s",3,[0.97,1.4],"minmax","OpenAI API · effort low · small coding task, 3 runs"],["cli-model-latency-tokens","cli-vs-api-small-coding-latency","First useful output","OpenAI API · GPT-6.1 Sol · high","cli-vs-api-small-coding-latency/First useful output","CLI vs API: time for a small coding task (First useful output)",5.31,"seconds","5.31 s",3,[4.99,6.42],"minmax","OpenAI API · effort high · small coding task, 3 runs"],["cli-model-latency-tokens","cli-vs-api-prompt-overhead","Input tokens","OpenAI API · GPT-6.1 Sol · low","cli-vs-api-prompt-overhead","Hidden prompt: input tokens for the same one-line request",17,"tokens","17",5,"\u0001","\u0001","OpenAI API · effort low · short fixed tasks"],["cli-model-latency-tokens","cli-vs-api-prompt-overhead","Input tokens","OpenAI API · GPT-6.1 Sol · high","cli-vs-api-prompt-overhead","Hidden prompt: input tokens for the same one-line request",17,"tokens","17",5,"\u0001","\u0001","OpenAI API · effort high · short fixed tasks"],["cli-model-latency-tokens","scheduler-repair-claude-vs-codex","Total time","OpenAI API · GPT-6.1 Sol · medium","scheduler-repair-claude-vs-codex/Total time","Repairing a scheduler: Claude Code vs Codex vs API (Total time)",17.32,"seconds","17.3 s",3,[16.28,18.61],"minmax","OpenAI API · effort medium · scheduler repair, 296 checks, 3 runs"],["cli-model-latency-tokens","scheduler-repair-claude-vs-codex","First useful output","OpenAI API · GPT-6.1 Sol · medium","scheduler-repair-claude-vs-codex/First useful output","Repairing a scheduler: Claude Code vs Codex vs API (First useful output)",7.46,"seconds","7.46 s",3,[6.68,9.05],"minmax","OpenAI API · effort medium · scheduler repair, 296 checks, 3 runs"],["cli-model-latency-tokens","scheduler-repair-output-tokens","Output tokens","OpenAI API · GPT-6.1 Sol · medium","scheduler-repair-output-tokens/Output tokens","Output tokens to repair the scheduler (Output tokens)",1313,"tokens","1,313",3,"\u0001","\u0001","OpenAI API · effort medium · scheduler repair, 296 checks, 3 runs"]]}],["gpt-6-luna-codex-cli","GPT-6 Luna (Codex CLI)","OpenAI","model","OpenAI’s GPT-6 Luna model run through the Codex CLI.",["GPT-6 Luna · Codex CLI"],{"$k":["studySlug","chartId","series","point","metric","label","value","unit","display","n","range","spanKind","context","ci","polarity","calculation"],"$r":[["cli-model-latency-tokens","cli-vs-api-exact-reply-latency","Total time","Codex CLI · GPT-6 Luna · none","cli-vs-api-exact-reply-latency/Total time","CLI vs API: time for a one-line answer (Total time)",3.19,"seconds","3.19 s",5,[2.88,3.83],"minmax","Codex CLI · effort none · fixed exact reply, 5 runs","\u0001","\u0001","\u0001"],["cli-model-latency-tokens","cli-vs-api-exact-reply-latency","First useful output","Codex CLI · GPT-6 Luna · none","cli-vs-api-exact-reply-latency/First useful output","CLI vs API: time for a one-line answer (First useful output)",2.79,"seconds","2.79 s",5,[2.46,3.42],"minmax","Codex CLI · effort none · fixed exact reply, 5 runs","\u0001","\u0001","\u0001"],["cli-model-latency-tokens","cli-vs-api-small-coding-latency","Total time","Codex CLI · GPT-6 Luna · none","cli-vs-api-small-coding-latency/Total time","CLI vs API: time for a small coding task (Total time)",9.23,"seconds","9.23 s",3,[8.99,11.68],"minmax","Codex CLI · effort none · small coding task, 3 runs","\u0001","\u0001","\u0001"],["cli-model-latency-tokens","cli-vs-api-small-coding-latency","First useful output","Codex CLI · GPT-6 Luna · none","cli-vs-api-small-coding-latency/First useful output","CLI vs API: time for a small coding task (First useful output)",8.68,"seconds","8.68 s",3,[8.27,11.01],"minmax","Codex CLI · effort none · small coding task, 3 runs","\u0001","\u0001","\u0001"],["cli-model-latency-tokens","cli-vs-api-prompt-overhead","Input tokens","Codex CLI · GPT-6 Luna · none","cli-vs-api-prompt-overhead","Hidden prompt: input tokens for the same one-line request",18859,"tokens","18,859",5,"\u0001","\u0001","Codex CLI · effort none · short fixed tasks","\u0001","\u0001","\u0001"],["single-call-vs-agent-loop","agent-loop-pass-rate","Strict pass","GPT-6 Luna (single call) · Codex CLI","agent-loop-pass-rate","Strict pass rate: single call vs agent loop on eight hard tasks",0.625,"rate","63% (10/16)",16,"\u0001","ci95","Codex CLI · single call",[0.3864,0.8152],"higher","\u0001"],["single-call-vs-agent-loop","agent-loop-pass-rate","Strict pass","GPT-6 Luna (agent loop) · Codex CLI","agent-loop-pass-rate","Strict pass rate: single call vs agent loop on eight hard tasks",0.8571,"rate","86% (12/14)",14,"\u0001","ci95","Codex CLI · agent loop",[0.6006,0.9599],"higher","\u0001"],["single-call-vs-agent-loop","agent-loop-by-task","GPT-6 Luna (single call) · Codex CLI","Interval merge fix","agent-loop-by-task/Interval merge fix","Strict passes per task: single call vs agent loop: Interval merge fix",1,"rate","100% (2/2)",2,"\u0001","ci95","Codex CLI · single call",[0.3424,1],"higher","\u0001"],["single-call-vs-agent-loop","agent-loop-by-task","GPT-6 Luna (single call) · Codex CLI","DST day-length fix","agent-loop-by-task/DST day-length fix","Strict passes per task: single call vs agent loop: DST day-length fix",1,"rate","100% (2/2)",2,"\u0001","ci95","Codex CLI · single call",[0.3424,1],"higher","\u0001"],["single-call-vs-agent-loop","agent-loop-by-task","GPT-6 Luna (single call) · Codex CLI","CSV parser","agent-loop-by-task/CSV parser","Strict passes per task: single call vs agent loop: CSV parser",1,"rate","100% (2/2)",2,"\u0001","ci95","Codex CLI · single call",[0.3424,1],"higher","\u0001"],["single-call-vs-agent-loop","agent-loop-by-task","GPT-6 Luna (single call) · Codex CLI","Event-loop order","agent-loop-by-task/Event-loop order","Strict passes per task: single call vs agent loop: Event-loop order",0,"rate","0% (0/2)",2,"\u0001","ci95","Codex CLI · single call",[0,0.6576],"higher","\u0001"],["single-call-vs-agent-loop","agent-loop-by-task","GPT-6 Luna (single call) · Codex CLI","Room schedule","agent-loop-by-task/Room schedule","Strict passes per task: single call vs agent loop: Room schedule",0.5,"rate","50% (1/2)",2,"\u0001","ci95","Codex CLI · single call",[0.0945,0.9055],"higher","\u0001"],["single-call-vs-agent-loop","agent-loop-by-task","GPT-6 Luna (single call) · Codex CLI","SemVer regex","agent-loop-by-task/SemVer regex","Strict passes per task: single call vs agent loop: SemVer regex",0.5,"rate","50% (1/2)",2,"\u0001","ci95","Codex CLI · single call",[0.0945,0.9055],"higher","\u0001"],["single-call-vs-agent-loop","agent-loop-by-task","GPT-6 Luna (single call) · Codex CLI","Money refactor","agent-loop-by-task/Money refactor","Strict passes per task: single call vs agent loop: Money refactor",0,"rate","0% (0/2)",2,"\u0001","ci95","Codex CLI · single call",[0,0.6576],"higher","\u0001"],["single-call-vs-agent-loop","agent-loop-by-task","GPT-6 Luna (single call) · Codex CLI","SQL report","agent-loop-by-task/SQL report","Strict passes per task: single call vs agent loop: SQL report",1,"rate","100% (2/2)",2,"\u0001","ci95","Codex CLI · single call",[0.3424,1],"higher","\u0001"],["single-call-vs-agent-loop","agent-loop-by-task","GPT-6 Luna (agent loop) · Codex CLI","Interval merge fix","agent-loop-by-task/Interval merge fix","Strict passes per task: single call vs agent loop: Interval merge fix",1,"rate","100% (2/2)",2,"\u0001","ci95","Codex CLI · agent loop",[0.3424,1],"higher","\u0001"],["single-call-vs-agent-loop","agent-loop-by-task","GPT-6 Luna (agent loop) · Codex CLI","CSV parser","agent-loop-by-task/CSV parser","Strict passes per task: single call vs agent loop: CSV parser",0.5,"rate","50% (1/2)",2,"\u0001","ci95","Codex CLI · agent loop",[0.0945,0.9055],"higher","\u0001"],["single-call-vs-agent-loop","agent-loop-by-task","GPT-6 Luna (agent loop) · Codex CLI","Event-loop order","agent-loop-by-task/Event-loop order","Strict passes per task: single call vs agent loop: Event-loop order",1,"rate","100% (2/2)",2,"\u0001","ci95","Codex CLI · agent loop",[0.3424,1],"higher","\u0001"],["single-call-vs-agent-loop","agent-loop-by-task","GPT-6 Luna (agent loop) · Codex CLI","Room schedule","agent-loop-by-task/Room schedule","Strict passes per task: single call vs agent loop: Room schedule",1,"rate","100% (2/2)",2,"\u0001","ci95","Codex CLI · agent loop",[0.3424,1],"higher","\u0001"],["single-call-vs-agent-loop","agent-loop-by-task","GPT-6 Luna (agent loop) · Codex CLI","SemVer regex","agent-loop-by-task/SemVer regex","Strict passes per task: single call vs agent loop: SemVer regex",1,"rate","100% (2/2)",2,"\u0001","ci95","Codex CLI · agent loop",[0.3424,1],"higher","\u0001"],["single-call-vs-agent-loop","agent-loop-by-task","GPT-6 Luna (agent loop) · Codex CLI","Money refactor","agent-loop-by-task/Money refactor","Strict passes per task: single call vs agent loop: Money refactor",0.5,"rate","50% (1/2)",2,"\u0001","ci95","Codex CLI · agent loop",[0.0945,0.9055],"higher","\u0001"],["single-call-vs-agent-loop","agent-loop-by-task","GPT-6 Luna (agent loop) · Codex CLI","SQL report","agent-loop-by-task/SQL report","Strict passes per task: single call vs agent loop: SQL report",1,"rate","100% (2/2)",2,"\u0001","ci95","Codex CLI · agent loop",[0.3424,1],"higher","\u0001"],["single-call-vs-agent-loop","agent-loop-total-time","Total time per attempt","GPT-6 Luna (single call) · Codex CLI","agent-loop-total-time","Total time per attempt: single call vs agent loop",5.16,"seconds","5.16 s",16,[3.59,11.32],"minmax","Codex CLI · single call","\u0001","\u0001","\u0001"],["single-call-vs-agent-loop","agent-loop-total-time","Total time per attempt","GPT-6 Luna (agent loop) · Codex CLI","agent-loop-total-time","Total time per attempt: single call vs agent loop",9.32,"seconds","9.32 s",14,[3.78,15.89],"minmax","Codex CLI · agent loop","\u0001","\u0001","\u0001"],["single-call-vs-agent-loop","agent-loop-tokens","Input tokens (cache reads included)","GPT-6 Luna (single call) · Codex CLI","agent-loop-tokens/Input tokens (cache reads included)","Tokens per attempt: single call vs agent loop (Input tokens (cache reads included))",11582,"tokens","11,582",16,[11526,11818],"minmax","Codex CLI · single call","\u0001","\u0001","\u0001"],["single-call-vs-agent-loop","agent-loop-tokens","Input tokens (cache reads included)","GPT-6 Luna (agent loop) · Codex CLI","agent-loop-tokens/Input tokens (cache reads included)","Tokens per attempt: single call vs agent loop (Input tokens (cache reads included))",15530,"tokens","15,530",14,[15391,39009],"minmax","Codex CLI · agent loop","\u0001","\u0001","\u0001"],["single-call-vs-agent-loop","agent-loop-tokens","Output tokens","GPT-6 Luna (single call) · Codex CLI","agent-loop-tokens/Output tokens","Tokens per attempt: single call vs agent loop (Output tokens)",345,"tokens","345",16,[36,634],"minmax","Codex CLI · single call","\u0001","\u0001","\u0001"],["single-call-vs-agent-loop","agent-loop-tokens","Output tokens","GPT-6 Luna (agent loop) · Codex CLI","agent-loop-tokens/Output tokens","Tokens per attempt: single call vs agent loop (Output tokens)",480,"tokens","480",14,[143,858],"minmax","Codex CLI · agent loop","\u0001","\u0001","\u0001"],["single-call-vs-agent-loop","agent-loop-tool-calls","Tool calls per attempt","GPT-6 Luna (agent loop) · Codex CLI","agent-loop-tool-calls","Tool calls per agent-loop attempt",0,"count","0",14,[0,1],"minmax","Codex CLI · agent loop","\u0001","\u0001","\u0001"],["single-call-vs-agent-loop","agent-loop-cost-per-pass","Cost per strict pass","GPT-6 Luna (single call) · Codex CLI","agent-loop-cost-per-pass","List-price cost per strict pass: single call vs agent loop (calculation)",0.00116,"usd","$0.0012",16,"\u0001","\u0001","Codex CLI · single call","\u0001","\u0001",true],["single-call-vs-agent-loop","agent-loop-cost-per-pass","Cost per strict pass","GPT-6 Luna (agent loop) · Codex CLI","agent-loop-cost-per-pass","List-price cost per strict pass: single call vs agent loop (calculation)",0.00099,"usd","$0.00099",14,"\u0001","\u0001","Codex CLI · agent loop","\u0001","\u0001",true],["llm-speed-anatomy","speed-anatomy-first-text","Time to first text","GPT-6 Luna (low) · Codex CLI","speed-anatomy-first-text","Time to first text: a 250-line answer, six models",3.3,"seconds","3.30 s",4,[3.19,3.47],"minmax","Codex CLI · effort low","\u0001","\u0001","\u0001"],["llm-speed-anatomy","speed-anatomy-output-speed","Visible tokens per second","GPT-6 Luna (low) · Codex CLI","speed-anatomy-output-speed","Output speed after the first text: visible tokens per second (calculation)",129.1,"tokens","129",4,[55.5,259.1],"minmax","Codex CLI · effort low","\u0001","\u0001",true],["llm-speed-anatomy","speed-anatomy-chars-per-second","Characters per second","GPT-6 Luna (low) · Codex CLI","speed-anatomy-chars-per-second","Output speed in characters per second after the first text (calculation)",524,"count","524",4,[225,1052],"minmax","Codex CLI · effort low","\u0001","\u0001",true]]}],["gpt-6-luna-openai-api","GPT-6 Luna (OpenAI API)","OpenAI","model","OpenAI’s GPT-6 Luna model called directly through the OpenAI API, without a CLI.",["GPT-6 Luna · OpenAI API"],{"$k":["studySlug","chartId","series","point","metric","label","value","unit","display","n","range","spanKind","context"],"$r":[["cli-model-latency-tokens","cli-vs-api-exact-reply-latency","Total time","OpenAI API · GPT-6 Luna · none","cli-vs-api-exact-reply-latency/Total time","CLI vs API: time for a one-line answer (Total time)",0.97,"seconds","0.97 s",5,[0.65,1.5],"minmax","OpenAI API · effort none · fixed exact reply, 5 runs"],["cli-model-latency-tokens","cli-vs-api-exact-reply-latency","First useful output","OpenAI API · GPT-6 Luna · none","cli-vs-api-exact-reply-latency/First useful output","CLI vs API: time for a one-line answer (First useful output)",0.82,"seconds","0.82 s",5,[0.51,1.37],"minmax","OpenAI API · effort none · fixed exact reply, 5 runs"],["cli-model-latency-tokens","cli-vs-api-small-coding-latency","Total time","OpenAI API · GPT-6 Luna · none","cli-vs-api-small-coding-latency/Total time","CLI vs API: time for a small coding task (Total time)",4.01,"seconds","4.01 s",3,[3.83,4.35],"minmax","OpenAI API · effort none · small coding task, 3 runs"],["cli-model-latency-tokens","cli-vs-api-small-coding-latency","First useful output","OpenAI API · GPT-6 Luna · none","cli-vs-api-small-coding-latency/First useful output","CLI vs API: time for a small coding task (First useful output)",0.67,"seconds","0.67 s",3,[0.62,0.81],"minmax","OpenAI API · effort none · small coding task, 3 runs"],["cli-model-latency-tokens","cli-vs-api-prompt-overhead","Input tokens","OpenAI API · GPT-6 Luna · none","cli-vs-api-prompt-overhead","Hidden prompt: input tokens for the same one-line request",17,"tokens","17",5,"\u0001","\u0001","OpenAI API · effort none · short fixed tasks"]]}],["gpt-5-2","GPT 5.2","OpenAI","model","OpenAI model in the public SWE-bench Verified panel (mini-SWE-agent v2, high effort).",["GPT 5.2"],{"$k":["studySlug","chartId","series","point","metric","label","value","unit","display","n","ci","spanKind","context"],"$r":[["swe-bench-verified","swebench-same-instance-leaderboard","Resolved rate","GPT 5.2 (high)","swebench-same-instance-leaderboard","Resolved rate on the same 33 SWE-bench Verified instances",0.8485,"rate","85% (28/33)",33,[0.6908,0.9335],"ci95","effort high · public mini-SWE-agent v2 run, same instances"],["swe-bench-verified","swebench-model-calls","Mean calls","GPT 5.2 (high)","swebench-model-calls","Model calls per instance",35.6,"calls","35.6",33,"\u0001","\u0001","effort high · public mini-SWE-agent v2 run, same instances"],["cost-thought-experiments","cost-per-resolved-agent-vs-panel","Cost per resolved instance","GPT 5.2 (high)","cost-per-resolved-agent-vs-panel","Recorded cost per resolved instance: Agent vs the public panel",0.628,"usd","$0.63",28,"\u0001","\u0001","effort high · public mini-SWE-agent v2 run, same instances"]]}],["gemini-3-flash","Gemini 3 Flash","Google","model","Google model in the public SWE-bench Verified panel (mini-SWE-agent v2, high effort).",["Gemini 3 Flash"],{"$k":["studySlug","chartId","series","point","metric","label","value","unit","display","n","ci","spanKind","context"],"$r":[["swe-bench-verified","swebench-same-instance-leaderboard","Resolved rate","Gemini 3 Flash (high)","swebench-same-instance-leaderboard","Resolved rate on the same 33 SWE-bench Verified instances",0.8182,"rate","82% (27/33)",33,[0.6561,0.9139],"ci95","effort high · public mini-SWE-agent v2 run, same instances"],["swe-bench-verified","swebench-model-calls","Mean calls","Gemini 3 Flash (high)","swebench-model-calls","Model calls per instance",54.2,"calls","54.2",33,"\u0001","\u0001","effort high · public mini-SWE-agent v2 run, same instances"],["cost-thought-experiments","cost-per-resolved-agent-vs-panel","Cost per resolved instance","Gemini 3 Flash (high)","cost-per-resolved-agent-vs-panel","Recorded cost per resolved instance: Agent vs the public panel",0.436,"usd","$0.44",27,"\u0001","\u0001","effort high · public mini-SWE-agent v2 run, same instances"]]}],["glm-5","GLM 5","Z.ai","model","Z.ai model in the public SWE-bench Verified panel (mini-SWE-agent v2, high effort).",["GLM 5"],{"$k":["studySlug","chartId","series","point","metric","label","value","unit","display","n","ci","spanKind","context"],"$r":[["swe-bench-verified","swebench-same-instance-leaderboard","Resolved rate","GLM 5 (high)","swebench-same-instance-leaderboard","Resolved rate on the same 33 SWE-bench Verified instances",0.7879,"rate","79% (26/33)",33,[0.6225,0.8932],"ci95","effort high · public mini-SWE-agent v2 run, same instances"],["swe-bench-verified","swebench-model-calls","Mean calls","GLM 5 (high)","swebench-model-calls","Model calls per instance",77.5,"calls","77.5",33,"\u0001","\u0001","effort high · public mini-SWE-agent v2 run, same instances"],["cost-thought-experiments","cost-per-resolved-agent-vs-panel","Cost per resolved instance","GLM 5 (high)","cost-per-resolved-agent-vs-panel","Recorded cost per resolved instance: Agent vs the public panel",0.667,"usd","$0.67",26,"\u0001","\u0001","effort high · public mini-SWE-agent v2 run, same instances"]]}],["claude-sonnet-4-5","Claude Sonnet 4.5","Anthropic","model","Anthropic model in the public SWE-bench Verified panel (mini-SWE-agent v2, high effort).",["Claude Sonnet 4.5","Claude 4.5 Sonnet"],{"$k":["studySlug","chartId","series","point","metric","label","value","unit","display","n","ci","spanKind","context"],"$r":[["swe-bench-verified","swebench-same-instance-leaderboard","Resolved rate","Claude 4.5 Sonnet (high)","swebench-same-instance-leaderboard","Resolved rate on the same 33 SWE-bench Verified instances",0.7576,"rate","76% (25/33)",33,[0.5898,0.8717],"ci95","effort high · public mini-SWE-agent v2 run, same instances"],["swe-bench-verified","swebench-model-calls","Mean calls","Claude 4.5 Sonnet (high)","swebench-model-calls","Model calls per instance",51,"calls","51",33,"\u0001","\u0001","effort high · public mini-SWE-agent v2 run, same instances"],["cost-thought-experiments","cost-per-resolved-agent-vs-panel","Cost per resolved instance","Claude 4.5 Sonnet (high)","cost-per-resolved-agent-vs-panel","Recorded cost per resolved instance: Agent vs the public panel",0.913,"usd","$0.91",25,"\u0001","\u0001","effort high · public mini-SWE-agent v2 run, same instances"]]}],["claude-opus-4-5","Claude Opus 4.5","Anthropic","model","Anthropic model in the public SWE-bench Verified panel (mini-SWE-agent v2, high effort).",["Claude Opus 4.5","Claude 4.5 Opus"],{"$k":["studySlug","chartId","series","point","metric","label","value","unit","display","n","ci","spanKind","context"],"$r":[["swe-bench-verified","swebench-same-instance-leaderboard","Resolved rate","Claude 4.5 Opus (high)","swebench-same-instance-leaderboard","Resolved rate on the same 33 SWE-bench Verified instances",0.7273,"rate","73% (24/33)",33,[0.5578,0.8493],"ci95","effort high · public mini-SWE-agent v2 run, same instances"],["swe-bench-verified","swebench-model-calls","Mean calls","Claude 4.5 Opus (high)","swebench-model-calls","Model calls per instance",35.9,"calls","35.9",33,"\u0001","\u0001","effort high · public mini-SWE-agent v2 run, same instances"],["cost-thought-experiments","cost-per-resolved-agent-vs-panel","Cost per resolved instance","Claude 4.5 Opus (high)","cost-per-resolved-agent-vs-panel","Recorded cost per resolved instance: Agent vs the public panel",1.184,"usd","$1.18",24,"\u0001","\u0001","effort high · public mini-SWE-agent v2 run, same instances"]]}],["claude-opus-4-6","Claude Opus 4.6","Anthropic","model","Anthropic model in the public SWE-bench Verified panel (mini-SWE-agent v2).",["Claude Opus 4.6","Claude 4.6 Opus"],{"$k":["studySlug","chartId","series","point","metric","label","value","unit","display","n","ci","spanKind","context"],"$r":[["swe-bench-verified","swebench-same-instance-leaderboard","Resolved rate","Claude 4.6 Opus","swebench-same-instance-leaderboard","Resolved rate on the same 33 SWE-bench Verified instances",0.697,"rate","70% (23/33)",33,[0.5266,0.8262],"ci95","public mini-SWE-agent v2 run, same instances"],["swe-bench-verified","swebench-model-calls","Mean calls","Claude 4.6 Opus","swebench-model-calls","Model calls per instance",28.9,"calls","28.9",33,"\u0001","\u0001","public mini-SWE-agent v2 run, same instances"],["cost-thought-experiments","cost-per-resolved-agent-vs-panel","Cost per resolved instance","Claude 4.6 Opus","cost-per-resolved-agent-vs-panel","Recorded cost per resolved instance: Agent vs the public panel",0.875,"usd","$0.88",23,"\u0001","\u0001","public mini-SWE-agent v2 run, same instances"]]}],["deepseek-v3-2","DeepSeek V3.2","DeepSeek","model","DeepSeek model in the public SWE-bench Verified panel (mini-SWE-agent v2, high effort).",["DeepSeek V3.2"],{"$k":["studySlug","chartId","series","point","metric","label","value","unit","display","n","ci","spanKind","context"],"$r":[["swe-bench-verified","swebench-same-instance-leaderboard","Resolved rate","DeepSeek V3.2 (high)","swebench-same-instance-leaderboard","Resolved rate on the same 33 SWE-bench Verified instances",0.7273,"rate","73% (24/33)",33,[0.5578,0.8493],"ci95","effort high · public mini-SWE-agent v2 run, same instances"],["swe-bench-verified","swebench-model-calls","Mean calls","DeepSeek V3.2 (high)","swebench-model-calls","Model calls per instance",88.2,"calls","88.2",33,"\u0001","\u0001","effort high · public mini-SWE-agent v2 run, same instances"],["cost-thought-experiments","cost-per-resolved-agent-vs-panel","Cost per resolved instance","DeepSeek V3.2 (high)","cost-per-resolved-agent-vs-panel","Recorded cost per resolved instance: Agent vs the public panel",0.637,"usd","$0.64",24,"\u0001","\u0001","effort high · public mini-SWE-agent v2 run, same instances"]]}],["minimax-m2-5","MiniMax M2.5","MiniMax","model","MiniMax model in the public SWE-bench Verified panel (mini-SWE-agent v2, high effort).",["MiniMax M2.5"],{"$k":["studySlug","chartId","series","point","metric","label","value","unit","display","n","ci","spanKind","context"],"$r":[["swe-bench-verified","swebench-same-instance-leaderboard","Resolved rate","MiniMax M2.5 (high)","swebench-same-instance-leaderboard","Resolved rate on the same 33 SWE-bench Verified instances",0.697,"rate","70% (23/33)",33,[0.5266,0.8262],"ci95","effort high · public mini-SWE-agent v2 run, same instances"],["swe-bench-verified","swebench-model-calls","Mean calls","MiniMax M2.5 (high)","swebench-model-calls","Model calls per instance",58.4,"calls","58.4",33,"\u0001","\u0001","effort high · public mini-SWE-agent v2 run, same instances"],["cost-thought-experiments","cost-per-resolved-agent-vs-panel","Cost per resolved instance","MiniMax M2.5 (high)","cost-per-resolved-agent-vs-panel","Recorded cost per resolved instance: Agent vs the public panel",0.107,"usd","$0.11",23,"\u0001","\u0001","effort high · public mini-SWE-agent v2 run, same instances"]]}],["kimi-k2-5","Kimi K2.5","Moonshot AI","model","Moonshot AI model in the public SWE-bench Verified panel (mini-SWE-agent v2, high effort).",["Kimi K2.5"],{"$k":["studySlug","chartId","series","point","metric","label","value","unit","display","n","ci","spanKind","context"],"$r":[["swe-bench-verified","swebench-same-instance-leaderboard","Resolved rate","Kimi K2.5 (high)","swebench-same-instance-leaderboard","Resolved rate on the same 33 SWE-bench Verified instances",0.697,"rate","70% (23/33)",33,[0.5266,0.8262],"ci95","effort high · public mini-SWE-agent v2 run, same instances"],["swe-bench-verified","swebench-model-calls","Mean calls","Kimi K2.5 (high)","swebench-model-calls","Model calls per instance",56.7,"calls","56.7",33,"\u0001","\u0001","effort high · public mini-SWE-agent v2 run, same instances"],["cost-thought-experiments","cost-per-resolved-agent-vs-panel","Cost per resolved instance","Kimi K2.5 (high)","cost-per-resolved-agent-vs-panel","Recorded cost per resolved instance: Agent vs the public panel",0.256,"usd","$0.26",23,"\u0001","\u0001","effort high · public mini-SWE-agent v2 run, same instances"]]}],["gpt-5-mini","GPT 5 mini","OpenAI","model","OpenAI model in the public SWE-bench Verified panel (mini-SWE-agent v2).",["GPT 5 mini"],{"$k":["studySlug","chartId","series","point","metric","label","value","unit","display","n","ci","spanKind","context"],"$r":[["swe-bench-verified","swebench-same-instance-leaderboard","Resolved rate","GPT 5 mini","swebench-same-instance-leaderboard","Resolved rate on the same 33 SWE-bench Verified instances",0.6364,"rate","64% (21/33)",33,[0.4662,0.7781],"ci95","public mini-SWE-agent v2 run, same instances"],["swe-bench-verified","swebench-model-calls","Mean calls","GPT 5 mini","swebench-model-calls","Model calls per instance",20.8,"calls","20.8",33,"\u0001","\u0001","public mini-SWE-agent v2 run, same instances"],["cost-thought-experiments","cost-per-resolved-agent-vs-panel","Cost per resolved instance","GPT 5 mini","cost-per-resolved-agent-vs-panel","Recorded cost per resolved instance: Agent vs the public panel",0.08,"usd","$0.080",21,"\u0001","\u0001","public mini-SWE-agent v2 run, same instances"]]}],["claude-opus-5","Claude Opus 5","Anthropic","model","Anthropic model, measured as a blind code-review critic and in a list-price calculation.",["Claude Opus 5"],[{"studySlug":"blind-review-head-to-head","chartId":"blind-review-critic-agreement","series":"Critic model","point":"Claude Opus 5 (Anthropic)","metric":"blind-review-critic-agreement","label":"Does the judge’s model family matter?","value":0.7,"unit":"rate","display":"70% (28/40)","n":40,"ci":[0.5457,0.8193],"spanKind":"ci95","context":"blind review panel"},{"studySlug":"cost-thought-experiments","chartId":"repriced-cost-per-resolved","series":"Repriced cost per resolved instance","point":"Claude Opus 5","metric":"repriced-cost-per-resolved","label":"Thought experiment: the same tokens at other list prices","value":8.723,"unit":"usd","display":"$8.72","calculation":true,"context":"calculation: Agent’s recorded tokens at this model’s list price"}]],["claude-fable-5","Claude Fable 5","Anthropic","model","Anthropic model, measured as a blind code-review critic.",["Claude Fable 5"],[{"studySlug":"blind-review-head-to-head","chartId":"blind-review-critic-agreement","series":"Critic model","point":"Claude Fable 5 (Anthropic)","metric":"blind-review-critic-agreement","label":"Does the judge’s model family matter?","value":0.65,"unit":"rate","display":"65% (26/40)","n":40,"ci":[0.4951,0.7787],"spanKind":"ci95","context":"blind review panel"}]],["gpt-5-5","GPT 5.5","OpenAI","model","OpenAI model, measured as a blind code-review critic on a small number of pairs.",["GPT 5.5"],[{"studySlug":"blind-review-head-to-head","chartId":"blind-review-critic-agreement","series":"Critic model","point":"GPT 5.5 (OpenAI)","metric":"blind-review-critic-agreement","label":"Does the judge’s model family matter?","value":1,"unit":"rate","display":"100% (2/2)","n":2,"ci":[0.3424,1],"spanKind":"ci95","context":"blind review panel"}]],["deterministic-routing-policy","Deterministic routing policy","Agent","router","Agent’s rule-based routing: an in-process policy picks the model and effort for each call from the task stage and signals. No model call, so no token cost.",["Deterministic routing policy"],{"$k":["studySlug","chartId","series","point","metric","label","value","unit","display","n","range","spanKind","context","ci","calculation"],"$r":[["routing-overhead","router-overhead-decision-latency","Decision time","Deterministic routing policy (Agent, in process)","router-overhead-decision-latency","Time to make one routing decision",0.00142,"ms","1.42 µs",20000,[0.00142,0.00233],"p50-p95","Agent · in process · routing overhead per decision","\u0001","\u0001"],["routing-overhead","router-overhead-completed","Completed","Deterministic routing policy (Agent, in process)","router-overhead-completed","Routing calls that returned a decision",1,"rate","100% (20000/20000)",20000,"\u0001","ci95","Agent · in process · routing overhead per decision",[0.9998,1],"\u0001"],["routing-overhead","router-overhead-cost-reported","Cost per 1,000 decisions","Deterministic routing policy (Agent, in process)","router-overhead-cost-reported","Cost per 1,000 routing decisions: no model call vs provider-reported",0,"usd","$0.00",20000,"\u0001","\u0001","Agent · in process · routing overhead per decision","\u0001","\u0001"],["routing-overhead","router-overhead-cost-per-1000-tasks","Every model call routed (49.5 per task)","Deterministic routing policy (Agent, in process)","router-overhead-cost-per-1000-tasks/Every model call routed (49.5 per task)","Added routing cost per 1,000 tasks (calculation) (Every model call routed (49.5 per task))",0,"usd","$0.00","\u0001","\u0001","\u0001","Agent · in process · calculation per 1,000 tasks from recorded decision counts","\u0001",true],["routing-overhead","router-overhead-cost-per-1000-tasks","Only System One decisions (7 per task)","Deterministic routing policy (Agent, in process)","router-overhead-cost-per-1000-tasks/Only System One decisions (7 per task)","Added routing cost per 1,000 tasks (calculation) (Only System One decisions (7 per task))",0,"usd","$0.00","\u0001","\u0001","\u0001","Agent · in process · calculation per 1,000 tasks from recorded decision counts","\u0001",true],["routing-overhead","router-overhead-delay-per-task","Every model call routed (49.5 per task)","Deterministic routing policy (Agent, in process)","router-overhead-delay-per-task/Every model call routed (49.5 per task)","Added routing delay per task (calculation) (Every model call routed (49.5 per task))",0.0000703,"seconds","70.3 µs","\u0001","\u0001","\u0001","Agent · in process · calculation per task from recorded decision counts, decisions in line","\u0001",true],["routing-overhead","router-overhead-delay-per-task","Only System One decisions (7 per task)","Deterministic routing policy (Agent, in process)","router-overhead-delay-per-task/Only System One decisions (7 per task)","Added routing delay per task (calculation) (Only System One decisions (7 per task))",0.0000099,"seconds","9.9 µs","\u0001","\u0001","\u0001","Agent · in process · calculation per task from recorded decision counts, decisions in line","\u0001",true]]}],["openrouter","OpenRouter","OpenRouter","provider","A gateway that routes one API to many inference providers. Its per-token price is compared with first-party list prices; its fee is charged when credits are bought.",["OpenRouter"],{"$k":["studySlug","chartId","series","point","metric","label","value","unit","display","context"],"$r":[["inference-provider-index","gateway-vs-direct-claude-haiku-4-5","Input","OpenRouter","gateway-vs-direct-claude-haiku-4-5/Input","Claude Haiku 4.5: OpenRouter vs Anthropic list price (Input)",1,"usd","$1.00","Claude Haiku 4.5 · list price, snapshot 2026-10-06"],["inference-provider-index","gateway-vs-direct-claude-haiku-4-5","Output","OpenRouter","gateway-vs-direct-claude-haiku-4-5/Output","Claude Haiku 4.5: OpenRouter vs Anthropic list price (Output)",5,"usd","$5.00","Claude Haiku 4.5 · list price, snapshot 2026-10-06"],["inference-provider-index","gateway-vs-direct-claude-sonnet-5","Input","OpenRouter","gateway-vs-direct-claude-sonnet-5/Input","Claude Sonnet 5: OpenRouter vs Anthropic list price (Input)",2,"usd","$2.00","Claude Sonnet 5 · list price, snapshot 2026-10-06"],["inference-provider-index","gateway-vs-direct-claude-sonnet-5","Output","OpenRouter","gateway-vs-direct-claude-sonnet-5/Output","Claude Sonnet 5: OpenRouter vs Anthropic list price (Output)",10,"usd","$10.00","Claude Sonnet 5 · list price, snapshot 2026-10-06"],["inference-provider-index","gateway-vs-direct-claude-sonnet-5-5","Input","OpenRouter","gateway-vs-direct-claude-sonnet-5-5/Input","Claude Sonnet 5.5: OpenRouter vs Anthropic list price (Input)",2,"usd","$2.00","Claude Sonnet 5.5 · list price, snapshot 2026-10-06"],["inference-provider-index","gateway-vs-direct-claude-sonnet-5-5","Output","OpenRouter","gateway-vs-direct-claude-sonnet-5-5/Output","Claude Sonnet 5.5: OpenRouter vs Anthropic list price (Output)",10,"usd","$10.00","Claude Sonnet 5.5 · list price, snapshot 2026-10-06"],["inference-provider-index","gateway-vs-direct-claude-opus-4-8","Input","OpenRouter","gateway-vs-direct-claude-opus-4-8/Input","Claude Opus 4.8: OpenRouter vs Anthropic list price (Input)",5,"usd","$5.00","Claude Opus 4.8 · list price, snapshot 2026-10-06"],["inference-provider-index","gateway-vs-direct-claude-opus-4-8","Output","OpenRouter","gateway-vs-direct-claude-opus-4-8/Output","Claude Opus 4.8: OpenRouter vs Anthropic list price (Output)",25,"usd","$25.00","Claude Opus 4.8 · list price, snapshot 2026-10-06"],["inference-provider-index","gateway-vs-direct-claude-opus-5","Input","OpenRouter","gateway-vs-direct-claude-opus-5/Input","Claude Opus 5: OpenRouter vs Anthropic list price (Input)",5,"usd","$5.00","Claude Opus 5 · list price, snapshot 2026-10-06"],["inference-provider-index","gateway-vs-direct-claude-opus-5","Output","OpenRouter","gateway-vs-direct-claude-opus-5/Output","Claude Opus 5: OpenRouter vs Anthropic list price (Output)",25,"usd","$25.00","Claude Opus 5 · list price, snapshot 2026-10-06"],["inference-provider-index","gateway-vs-direct-claude-opus-5-5","Input","OpenRouter","gateway-vs-direct-claude-opus-5-5/Input","Claude Opus 5.5: OpenRouter vs Anthropic list price (Input)",4,"usd","$4.00","Claude Opus 5.5 · list price, snapshot 2026-10-06"],["inference-provider-index","gateway-vs-direct-claude-opus-5-5","Output","OpenRouter","gateway-vs-direct-claude-opus-5-5/Output","Claude Opus 5.5: OpenRouter vs Anthropic list price (Output)",20,"usd","$20.00","Claude Opus 5.5 · list price, snapshot 2026-10-06"],["inference-provider-index","gateway-vs-direct-claude-fable-5-1","Input","OpenRouter","gateway-vs-direct-claude-fable-5-1/Input","Claude Fable 5.1: OpenRouter vs Anthropic list price (Input)",10,"usd","$10.00","Claude Fable 5.1 · list price, snapshot 2026-10-06"],["inference-provider-index","gateway-vs-direct-claude-fable-5-1","Output","OpenRouter","gateway-vs-direct-claude-fable-5-1/Output","Claude Fable 5.1: OpenRouter vs Anthropic list price (Output)",50,"usd","$50.00","Claude Fable 5.1 · list price, snapshot 2026-10-06"],["inference-provider-index","gateway-vs-direct-gpt-6-luna","Input","OpenRouter","gateway-vs-direct-gpt-6-luna/Input","GPT-6 Luna: OpenRouter vs OpenAI list price (Input)",0.1,"usd","$0.10","GPT-6 Luna · list price, snapshot 2026-10-06"],["inference-provider-index","gateway-vs-direct-gpt-6-luna","Output","OpenRouter","gateway-vs-direct-gpt-6-luna/Output","GPT-6 Luna: OpenRouter vs OpenAI list price (Output)",0.5,"usd","$0.50","GPT-6 Luna · list price, snapshot 2026-10-06"],["inference-provider-index","gateway-vs-direct-gemini-3-8-flash","Input","OpenRouter","gateway-vs-direct-gemini-3-8-flash/Input","Gemini 3.8 Flash: OpenRouter vs Google list price (Input)",0.75,"usd","$0.75","Gemini 3.8 Flash · list price, snapshot 2026-10-06"],["inference-provider-index","gateway-vs-direct-gemini-3-8-flash","Output","OpenRouter","gateway-vs-direct-gemini-3-8-flash/Output","Gemini 3.8 Flash: OpenRouter vs Google list price (Output)",3.75,"usd","$3.75","Gemini 3.8 Flash · list price, snapshot 2026-10-06"],["inference-provider-index","gateway-vs-direct-gemini-3-5-flash","Input","OpenRouter","gateway-vs-direct-gemini-3-5-flash/Input","Gemini 3.5 Flash: OpenRouter vs Google list price (Input)",1.5,"usd","$1.50","Gemini 3.5 Flash · list price, snapshot 2026-10-06"],["inference-provider-index","gateway-vs-direct-gemini-3-5-flash","Output","OpenRouter","gateway-vs-direct-gemini-3-5-flash/Output","Gemini 3.5 Flash: OpenRouter vs Google list price (Output)",9,"usd","$9.00","Gemini 3.5 Flash · list price, snapshot 2026-10-06"]]}],["anthropic","Anthropic","Anthropic","provider","Anthropic’s own API: the first-party list price of the Claude models, and Anthropic’s endpoint as listed on OpenRouter.",["Anthropic"],{"$k":["studySlug","chartId","series","point","metric","label","value","unit","display","context"],"$r":[["inference-provider-index","gateway-vs-direct-claude-haiku-4-5","Input","Anthropic (first-party list price)","gateway-vs-direct-claude-haiku-4-5/Input","Claude Haiku 4.5: OpenRouter vs Anthropic list price (Input)",1,"usd","$1.00","first-party list price · Claude Haiku 4.5 · list price, snapshot 2026-10-06"],["inference-provider-index","gateway-vs-direct-claude-haiku-4-5","Output","Anthropic (first-party list price)","gateway-vs-direct-claude-haiku-4-5/Output","Claude Haiku 4.5: OpenRouter vs Anthropic list price (Output)",5,"usd","$5.00","first-party list price · Claude Haiku 4.5 · list price, snapshot 2026-10-06"],["inference-provider-index","gateway-vs-direct-claude-sonnet-5","Input","Anthropic (first-party list price)","gateway-vs-direct-claude-sonnet-5/Input","Claude Sonnet 5: OpenRouter vs Anthropic list price (Input)",2,"usd","$2.00","first-party list price · Claude Sonnet 5 · list price, snapshot 2026-10-06"],["inference-provider-index","gateway-vs-direct-claude-sonnet-5","Output","Anthropic (first-party list price)","gateway-vs-direct-claude-sonnet-5/Output","Claude Sonnet 5: OpenRouter vs Anthropic list price (Output)",10,"usd","$10.00","first-party list price · Claude Sonnet 5 · list price, snapshot 2026-10-06"],["inference-provider-index","gateway-vs-direct-claude-sonnet-5-5","Input","Anthropic (first-party list price)","gateway-vs-direct-claude-sonnet-5-5/Input","Claude Sonnet 5.5: OpenRouter vs Anthropic list price (Input)",2,"usd","$2.00","first-party list price · Claude Sonnet 5.5 · list price, snapshot 2026-10-06"],["inference-provider-index","gateway-vs-direct-claude-sonnet-5-5","Output","Anthropic (first-party list price)","gateway-vs-direct-claude-sonnet-5-5/Output","Claude Sonnet 5.5: OpenRouter vs Anthropic list price (Output)",10,"usd","$10.00","first-party list price · Claude Sonnet 5.5 · list price, snapshot 2026-10-06"],["inference-provider-index","gateway-vs-direct-claude-opus-4-8","Input","Anthropic (first-party list price)","gateway-vs-direct-claude-opus-4-8/Input","Claude Opus 4.8: OpenRouter vs Anthropic list price (Input)",5,"usd","$5.00","first-party list price · Claude Opus 4.8 · list price, snapshot 2026-10-06"],["inference-provider-index","gateway-vs-direct-claude-opus-4-8","Output","Anthropic (first-party list price)","gateway-vs-direct-claude-opus-4-8/Output","Claude Opus 4.8: OpenRouter vs Anthropic list price (Output)",25,"usd","$25.00","first-party list price · Claude Opus 4.8 · list price, snapshot 2026-10-06"],["inference-provider-index","gateway-vs-direct-claude-opus-5","Input","Anthropic (first-party list price)","gateway-vs-direct-claude-opus-5/Input","Claude Opus 5: OpenRouter vs Anthropic list price (Input)",5,"usd","$5.00","first-party list price · Claude Opus 5 · list price, snapshot 2026-10-06"],["inference-provider-index","gateway-vs-direct-claude-opus-5","Output","Anthropic (first-party list price)","gateway-vs-direct-claude-opus-5/Output","Claude Opus 5: OpenRouter vs Anthropic list price (Output)",25,"usd","$25.00","first-party list price · Claude Opus 5 · list price, snapshot 2026-10-06"],["inference-provider-index","gateway-vs-direct-claude-opus-5-5","Input","Anthropic (first-party list price)","gateway-vs-direct-claude-opus-5-5/Input","Claude Opus 5.5: OpenRouter vs Anthropic list price (Input)",4,"usd","$4.00","first-party list price · Claude Opus 5.5 · list price, snapshot 2026-10-06"],["inference-provider-index","gateway-vs-direct-claude-opus-5-5","Output","Anthropic (first-party list price)","gateway-vs-direct-claude-opus-5-5/Output","Claude Opus 5.5: OpenRouter vs Anthropic list price (Output)",20,"usd","$20.00","first-party list price · Claude Opus 5.5 · list price, snapshot 2026-10-06"],["inference-provider-index","gateway-vs-direct-claude-fable-5-1","Input","Anthropic (first-party list price)","gateway-vs-direct-claude-fable-5-1/Input","Claude Fable 5.1: OpenRouter vs Anthropic list price (Input)",10,"usd","$10.00","first-party list price · Claude Fable 5.1 · list price, snapshot 2026-10-06"],["inference-provider-index","gateway-vs-direct-claude-fable-5-1","Output","Anthropic (first-party list price)","gateway-vs-direct-claude-fable-5-1/Output","Claude Fable 5.1: OpenRouter vs Anthropic list price (Output)",50,"usd","$50.00","first-party list price · Claude Fable 5.1 · list price, snapshot 2026-10-06"],["inference-provider-index","provider-prices-claude-haiku-4-5","Input","Anthropic","provider-prices-claude-haiku-4-5/Input","Claude Haiku 4.5: price per million tokens by provider (Input)",1,"usd","$1.00","Claude Haiku 4.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-claude-haiku-4-5","Output","Anthropic","provider-prices-claude-haiku-4-5/Output","Claude Haiku 4.5: price per million tokens by provider (Output)",5,"usd","$5.00","Claude Haiku 4.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-claude-haiku-4-5","Cache read","Anthropic","provider-prices-claude-haiku-4-5/Cache read","Claude Haiku 4.5: price per million tokens by provider (Cache read)",0.1,"usd","$0.10","Claude Haiku 4.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-claude-sonnet-5","Input","Anthropic","provider-prices-claude-sonnet-5/Input","Claude Sonnet 5: price per million tokens by provider (Input)",2,"usd","$2.00","Claude Sonnet 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-claude-sonnet-5","Output","Anthropic","provider-prices-claude-sonnet-5/Output","Claude Sonnet 5: price per million tokens by provider (Output)",10,"usd","$10.00","Claude Sonnet 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-claude-sonnet-5","Cache read","Anthropic","provider-prices-claude-sonnet-5/Cache read","Claude Sonnet 5: price per million tokens by provider (Cache read)",0.2,"usd","$0.20","Claude Sonnet 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-claude-sonnet-5-5","Input","Anthropic","provider-prices-claude-sonnet-5-5/Input","Claude Sonnet 5.5: price per million tokens by provider (Input)",2,"usd","$2.00","Claude Sonnet 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-claude-sonnet-5-5","Output","Anthropic","provider-prices-claude-sonnet-5-5/Output","Claude Sonnet 5.5: price per million tokens by provider (Output)",10,"usd","$10.00","Claude Sonnet 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-claude-sonnet-5-5","Cache read","Anthropic","provider-prices-claude-sonnet-5-5/Cache read","Claude Sonnet 5.5: price per million tokens by provider (Cache read)",0.2,"usd","$0.20","Claude Sonnet 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-claude-opus-4-8","Input","Anthropic","provider-prices-claude-opus-4-8/Input","Claude Opus 4.8: price per million tokens by provider (Input)",5,"usd","$5.00","Claude Opus 4.8 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-claude-opus-4-8","Output","Anthropic","provider-prices-claude-opus-4-8/Output","Claude Opus 4.8: price per million tokens by provider (Output)",25,"usd","$25.00","Claude Opus 4.8 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-claude-opus-4-8","Cache read","Anthropic","provider-prices-claude-opus-4-8/Cache read","Claude Opus 4.8: price per million tokens by provider (Cache read)",0.5,"usd","$0.50","Claude Opus 4.8 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-claude-opus-5","Input","Anthropic","provider-prices-claude-opus-5/Input","Claude Opus 5: price per million tokens by provider (Input)",5,"usd","$5.00","Claude Opus 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-claude-opus-5","Output","Anthropic","provider-prices-claude-opus-5/Output","Claude Opus 5: price per million tokens by provider (Output)",25,"usd","$25.00","Claude Opus 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-claude-opus-5","Cache read","Anthropic","provider-prices-claude-opus-5/Cache read","Claude Opus 5: price per million tokens by provider (Cache read)",0.5,"usd","$0.50","Claude Opus 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-claude-opus-5-5","Input","Anthropic","provider-prices-claude-opus-5-5/Input","Claude Opus 5.5: price per million tokens by provider (Input)",4,"usd","$4.00","Claude Opus 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-claude-opus-5-5","Output","Anthropic","provider-prices-claude-opus-5-5/Output","Claude Opus 5.5: price per million tokens by provider (Output)",20,"usd","$20.00","Claude Opus 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-claude-opus-5-5","Cache read","Anthropic","provider-prices-claude-opus-5-5/Cache read","Claude Opus 5.5: price per million tokens by provider (Cache read)",0.2,"usd","$0.20","Claude Opus 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-claude-fable-5-1","Input","Anthropic","provider-prices-claude-fable-5-1/Input","Claude Fable 5.1: price per million tokens by provider (Input)",10,"usd","$10.00","Claude Fable 5.1 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-claude-fable-5-1","Output","Anthropic","provider-prices-claude-fable-5-1/Output","Claude Fable 5.1: price per million tokens by provider (Output)",50,"usd","$50.00","Claude Fable 5.1 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-claude-fable-5-1","Cache read","Anthropic","provider-prices-claude-fable-5-1/Cache read","Claude Fable 5.1: price per million tokens by provider (Cache read)",0.25,"usd","$0.25","Claude Fable 5.1 · reported by OpenRouter’s public API, snapshot 2026-10-06"]]}],["openai","OpenAI","OpenAI","provider","OpenAI’s own API: the first-party list price of GPT models, and OpenAI’s endpoints as listed on OpenRouter.",["OpenAI"],{"$k":["studySlug","chartId","series","point","metric","label","value","unit","display","context"],"$r":[["inference-provider-index","gateway-vs-direct-gpt-6-luna","Input","OpenAI (first-party list price)","gateway-vs-direct-gpt-6-luna/Input","GPT-6 Luna: OpenRouter vs OpenAI list price (Input)",0.1,"usd","$0.10","first-party list price · GPT-6 Luna · list price, snapshot 2026-10-06"],["inference-provider-index","gateway-vs-direct-gpt-6-luna","Output","OpenAI (first-party list price)","gateway-vs-direct-gpt-6-luna/Output","GPT-6 Luna: OpenRouter vs OpenAI list price (Output)",0.5,"usd","$0.50","first-party list price · GPT-6 Luna · list price, snapshot 2026-10-06"],["inference-provider-index","provider-prices-gpt-6-sol","Input","OpenAI","provider-prices-gpt-6-sol/Input","GPT-6 Sol: price per million tokens by provider (Input)",2,"usd","$2.00","GPT-6 Sol · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-gpt-6-sol","Output","OpenAI","provider-prices-gpt-6-sol/Output","GPT-6 Sol: price per million tokens by provider (Output)",10,"usd","$10.00","GPT-6 Sol · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-gpt-6-sol","Cache read","OpenAI","provider-prices-gpt-6-sol/Cache read","GPT-6 Sol: price per million tokens by provider (Cache read)",0.2,"usd","$0.20","GPT-6 Sol · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-gpt-6-luna","Input","OpenAI","provider-prices-gpt-6-luna/Input","GPT-6 Luna: price per million tokens by provider (Input)",0.1,"usd","$0.10","GPT-6 Luna · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-gpt-6-luna","Output","OpenAI","provider-prices-gpt-6-luna/Output","GPT-6 Luna: price per million tokens by provider (Output)",0.5,"usd","$0.50","GPT-6 Luna · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-gpt-6-luna","Cache read","OpenAI","provider-prices-gpt-6-luna/Cache read","GPT-6 Luna: price per million tokens by provider (Cache read)",0.01,"usd","$0.010","GPT-6 Luna · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-gpt-6-astra","Input","OpenAI","provider-prices-gpt-6-astra/Input","GPT-6 Astra: price per million tokens by provider (Input)",10,"usd","$10.00","GPT-6 Astra · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-gpt-6-astra","Output","OpenAI","provider-prices-gpt-6-astra/Output","GPT-6 Astra: price per million tokens by provider (Output)",50,"usd","$50.00","GPT-6 Astra · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-gpt-6-astra","Cache read","OpenAI","provider-prices-gpt-6-astra/Cache read","GPT-6 Astra: price per million tokens by provider (Cache read)",1,"usd","$1.00","GPT-6 Astra · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-gpt-5-5","Input","OpenAI","provider-prices-gpt-5-5/Input","GPT-5.5: price per million tokens by provider (Input)",5,"usd","$5.00","GPT-5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-gpt-5-5","Output","OpenAI","provider-prices-gpt-5-5/Output","GPT-5.5: price per million tokens by provider (Output)",30,"usd","$30.00","GPT-5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-gpt-5-5","Cache read","OpenAI","provider-prices-gpt-5-5/Cache read","GPT-5.5: price per million tokens by provider (Cache read)",0.5,"usd","$0.50","GPT-5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"]]}],["google-ai-studio","Google AI Studio","Google","provider","Google’s Gemini API (AI Studio): the first-party list price of Gemini models, and its endpoints as listed on OpenRouter.",["Google AI Studio"],{"$k":["studySlug","chartId","series","point","metric","label","value","unit","display","context"],"$r":[["inference-provider-index","gateway-vs-direct-gemini-3-8-flash","Input","Google AI Studio (first-party list price)","gateway-vs-direct-gemini-3-8-flash/Input","Gemini 3.8 Flash: OpenRouter vs Google list price (Input)",0.75,"usd","$0.75","first-party list price · Gemini 3.8 Flash · list price, snapshot 2026-10-06"],["inference-provider-index","gateway-vs-direct-gemini-3-8-flash","Output","Google AI Studio (first-party list price)","gateway-vs-direct-gemini-3-8-flash/Output","Gemini 3.8 Flash: OpenRouter vs Google list price (Output)",3.75,"usd","$3.75","first-party list price · Gemini 3.8 Flash · list price, snapshot 2026-10-06"],["inference-provider-index","gateway-vs-direct-gemini-3-5-flash","Input","Google AI Studio (first-party list price)","gateway-vs-direct-gemini-3-5-flash/Input","Gemini 3.5 Flash: OpenRouter vs Google list price (Input)",1.5,"usd","$1.50","first-party list price · Gemini 3.5 Flash · list price, snapshot 2026-10-06"],["inference-provider-index","gateway-vs-direct-gemini-3-5-flash","Output","Google AI Studio (first-party list price)","gateway-vs-direct-gemini-3-5-flash/Output","Gemini 3.5 Flash: OpenRouter vs Google list price (Output)",9,"usd","$9.00","first-party list price · Gemini 3.5 Flash · list price, snapshot 2026-10-06"],["inference-provider-index","provider-prices-gemini-3-8-flash","Input","Google AI Studio","provider-prices-gemini-3-8-flash/Input","Gemini 3.8 Flash: price per million tokens by provider (Input)",0.75,"usd","$0.75","Gemini 3.8 Flash · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-gemini-3-8-flash","Output","Google AI Studio","provider-prices-gemini-3-8-flash/Output","Gemini 3.8 Flash: price per million tokens by provider (Output)",3.75,"usd","$3.75","Gemini 3.8 Flash · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-gemini-3-8-flash","Cache read","Google AI Studio","provider-prices-gemini-3-8-flash/Cache read","Gemini 3.8 Flash: price per million tokens by provider (Cache read)",0.075,"usd","$0.075","Gemini 3.8 Flash · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-gemini-3-5-flash","Input","Google AI Studio","provider-prices-gemini-3-5-flash/Input","Gemini 3.5 Flash: price per million tokens by provider (Input)",1.5,"usd","$1.50","Gemini 3.5 Flash · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-gemini-3-5-flash","Output","Google AI Studio","provider-prices-gemini-3-5-flash/Output","Gemini 3.5 Flash: price per million tokens by provider (Output)",9,"usd","$9.00","Gemini 3.5 Flash · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-gemini-3-5-flash","Cache read","Google AI Studio","provider-prices-gemini-3-5-flash/Cache read","Gemini 3.5 Flash: price per million tokens by provider (Cache read)",0.15,"usd","$0.15","Gemini 3.5 Flash · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-gemini-3-5-flash-lite","Input","Google AI Studio","provider-prices-gemini-3-5-flash-lite/Input","Gemini 3.5 Flash Lite: price per million tokens by provider (Input)",0.3,"usd","$0.30","Gemini 3.5 Flash Lite · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-gemini-3-5-flash-lite","Output","Google AI Studio","provider-prices-gemini-3-5-flash-lite/Output","Gemini 3.5 Flash Lite: price per million tokens by provider (Output)",2.5,"usd","$2.50","Gemini 3.5 Flash Lite · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-gemini-3-5-flash-lite","Cache read","Google AI Studio","provider-prices-gemini-3-5-flash-lite/Cache read","Gemini 3.5 Flash Lite: price per million tokens by provider (Cache read)",0.03,"usd","$0.030","Gemini 3.5 Flash Lite · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-gemini-3-1-pro-preview","Input","Google AI Studio","provider-prices-gemini-3-1-pro-preview/Input","Gemini 3.1 Pro Preview: price per million tokens by provider (Input)",2,"usd","$2.00","Gemini 3.1 Pro Preview · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-gemini-3-1-pro-preview","Output","Google AI Studio","provider-prices-gemini-3-1-pro-preview/Output","Gemini 3.1 Pro Preview: price per million tokens by provider (Output)",12,"usd","$12.00","Gemini 3.1 Pro Preview · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-gemini-3-1-pro-preview","Cache read","Google AI Studio","provider-prices-gemini-3-1-pro-preview/Cache read","Gemini 3.1 Pro Preview: price per million tokens by provider (Cache read)",0.2,"usd","$0.20","Gemini 3.1 Pro Preview · reported by OpenRouter’s public API, snapshot 2026-10-06"]]}],["google-vertex","Google Vertex AI","Google","provider","Google Cloud’s model platform. An inference provider listed on OpenRouter. Its prices here are the ones OpenRouter’s public API reported for its endpoints.",["Google Vertex"],{"$k":["studySlug","chartId","series","point","metric","label","value","unit","display","context"],"$r":[["inference-provider-index","provider-prices-claude-haiku-4-5","Input","Google Vertex","provider-prices-claude-haiku-4-5/Input","Claude Haiku 4.5: price per million tokens by provider (Input)",1,"usd","$1.00","Claude Haiku 4.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-claude-haiku-4-5","Output","Google Vertex","provider-prices-claude-haiku-4-5/Output","Claude Haiku 4.5: price per million tokens by provider (Output)",5,"usd","$5.00","Claude Haiku 4.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-claude-haiku-4-5","Cache read","Google Vertex","provider-prices-claude-haiku-4-5/Cache read","Claude Haiku 4.5: price per million tokens by provider (Cache read)",0.1,"usd","$0.10","Claude Haiku 4.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-claude-sonnet-5","Input","Google Vertex","provider-prices-claude-sonnet-5/Input","Claude Sonnet 5: price per million tokens by provider (Input)",2,"usd","$2.00","Claude Sonnet 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-claude-sonnet-5","Output","Google Vertex","provider-prices-claude-sonnet-5/Output","Claude Sonnet 5: price per million tokens by provider (Output)",10,"usd","$10.00","Claude Sonnet 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-claude-sonnet-5","Cache read","Google Vertex","provider-prices-claude-sonnet-5/Cache read","Claude Sonnet 5: price per million tokens by provider (Cache read)",0.2,"usd","$0.20","Claude Sonnet 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-claude-sonnet-5-5","Input","Google Vertex","provider-prices-claude-sonnet-5-5/Input","Claude Sonnet 5.5: price per million tokens by provider (Input)",2,"usd","$2.00","Claude Sonnet 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-claude-sonnet-5-5","Output","Google Vertex","provider-prices-claude-sonnet-5-5/Output","Claude Sonnet 5.5: price per million tokens by provider (Output)",10,"usd","$10.00","Claude Sonnet 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-claude-sonnet-5-5","Cache read","Google Vertex","provider-prices-claude-sonnet-5-5/Cache read","Claude Sonnet 5.5: price per million tokens by provider (Cache read)",0.2,"usd","$0.20","Claude Sonnet 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-claude-opus-4-8","Input","Google Vertex","provider-prices-claude-opus-4-8/Input","Claude Opus 4.8: price per million tokens by provider (Input)",5,"usd","$5.00","Claude Opus 4.8 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-claude-opus-4-8","Output","Google Vertex","provider-prices-claude-opus-4-8/Output","Claude Opus 4.8: price per million tokens by provider (Output)",25,"usd","$25.00","Claude Opus 4.8 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-claude-opus-4-8","Cache read","Google Vertex","provider-prices-claude-opus-4-8/Cache read","Claude Opus 4.8: price per million tokens by provider (Cache read)",0.5,"usd","$0.50","Claude Opus 4.8 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-claude-opus-5","Input","Google Vertex","provider-prices-claude-opus-5/Input","Claude Opus 5: price per million tokens by provider (Input)",5,"usd","$5.00","Claude Opus 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-claude-opus-5","Output","Google Vertex","provider-prices-claude-opus-5/Output","Claude Opus 5: price per million tokens by provider (Output)",25,"usd","$25.00","Claude Opus 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-claude-opus-5","Cache read","Google Vertex","provider-prices-claude-opus-5/Cache read","Claude Opus 5: price per million tokens by provider (Cache read)",0.5,"usd","$0.50","Claude Opus 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-claude-opus-5-5","Input","Google Vertex","provider-prices-claude-opus-5-5/Input","Claude Opus 5.5: price per million tokens by provider (Input)",4,"usd","$4.00","Claude Opus 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-claude-opus-5-5","Output","Google Vertex","provider-prices-claude-opus-5-5/Output","Claude Opus 5.5: price per million tokens by provider (Output)",20,"usd","$20.00","Claude Opus 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-claude-opus-5-5","Cache read","Google Vertex","provider-prices-claude-opus-5-5/Cache read","Claude Opus 5.5: price per million tokens by provider (Cache read)",0.2,"usd","$0.20","Claude Opus 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-claude-fable-5-1","Input","Google Vertex","provider-prices-claude-fable-5-1/Input","Claude Fable 5.1: price per million tokens by provider (Input)",10,"usd","$10.00","Claude Fable 5.1 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-claude-fable-5-1","Output","Google Vertex","provider-prices-claude-fable-5-1/Output","Claude Fable 5.1: price per million tokens by provider (Output)",50,"usd","$50.00","Claude Fable 5.1 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-claude-fable-5-1","Cache read","Google Vertex","provider-prices-claude-fable-5-1/Cache read","Claude Fable 5.1: price per million tokens by provider (Cache read)",0.25,"usd","$0.25","Claude Fable 5.1 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-gpt-oss-120b","Input","Google Vertex","provider-prices-gpt-oss-120b/Input","gpt-oss-120b: price per million tokens by provider (Input)",0.09,"usd","$0.090","gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-gpt-oss-120b","Output","Google Vertex","provider-prices-gpt-oss-120b/Output","gpt-oss-120b: price per million tokens by provider (Output)",0.36,"usd","$0.36","gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-gemini-3-8-flash","Input","Google Vertex","provider-prices-gemini-3-8-flash/Input","Gemini 3.8 Flash: price per million tokens by provider (Input)",0.75,"usd","$0.75","Gemini 3.8 Flash · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-gemini-3-8-flash","Output","Google Vertex","provider-prices-gemini-3-8-flash/Output","Gemini 3.8 Flash: price per million tokens by provider (Output)",3.75,"usd","$3.75","Gemini 3.8 Flash · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-gemini-3-8-flash","Cache read","Google Vertex","provider-prices-gemini-3-8-flash/Cache read","Gemini 3.8 Flash: price per million tokens by provider (Cache read)",0.075,"usd","$0.075","Gemini 3.8 Flash · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-gemini-3-5-flash","Input","Google Vertex","provider-prices-gemini-3-5-flash/Input","Gemini 3.5 Flash: price per million tokens by provider (Input)",1.5,"usd","$1.50","Gemini 3.5 Flash · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-gemini-3-5-flash","Output","Google Vertex","provider-prices-gemini-3-5-flash/Output","Gemini 3.5 Flash: price per million tokens by provider (Output)",9,"usd","$9.00","Gemini 3.5 Flash · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-gemini-3-5-flash","Cache read","Google Vertex","provider-prices-gemini-3-5-flash/Cache read","Gemini 3.5 Flash: price per million tokens by provider (Cache read)",0.15,"usd","$0.15","Gemini 3.5 Flash · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-gemini-3-5-flash-lite","Input","Google Vertex","provider-prices-gemini-3-5-flash-lite/Input","Gemini 3.5 Flash Lite: price per million tokens by provider (Input)",0.3,"usd","$0.30","Gemini 3.5 Flash Lite · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-gemini-3-5-flash-lite","Output","Google Vertex","provider-prices-gemini-3-5-flash-lite/Output","Gemini 3.5 Flash Lite: price per million tokens by provider (Output)",2.5,"usd","$2.50","Gemini 3.5 Flash Lite · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-gemini-3-5-flash-lite","Cache read","Google Vertex","provider-prices-gemini-3-5-flash-lite/Cache read","Gemini 3.5 Flash Lite: price per million tokens by provider (Cache read)",0.03,"usd","$0.030","Gemini 3.5 Flash Lite · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-gemini-3-1-pro-preview","Input","Google Vertex","provider-prices-gemini-3-1-pro-preview/Input","Gemini 3.1 Pro Preview: price per million tokens by provider (Input)",2,"usd","$2.00","Gemini 3.1 Pro Preview · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-gemini-3-1-pro-preview","Output","Google Vertex","provider-prices-gemini-3-1-pro-preview/Output","Gemini 3.1 Pro Preview: price per million tokens by provider (Output)",12,"usd","$12.00","Gemini 3.1 Pro Preview · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-gemini-3-1-pro-preview","Cache read","Google Vertex","provider-prices-gemini-3-1-pro-preview/Cache read","Gemini 3.1 Pro Preview: price per million tokens by provider (Cache read)",0.2,"usd","$0.20","Gemini 3.1 Pro Preview · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-llama-3-3-70b-instruct","Input","Google Vertex","provider-prices-llama-3-3-70b-instruct/Input","Llama 3.3 70B Instruct: price per million tokens by provider (Input)",0.72,"usd","$0.72","Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-llama-3-3-70b-instruct","Output","Google Vertex","provider-prices-llama-3-3-70b-instruct/Output","Llama 3.3 70B Instruct: price per million tokens by provider (Output)",0.72,"usd","$0.72","Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06"]]}],["amazon-bedrock","Amazon Bedrock","Amazon","provider","Amazon Web Services’ model platform. An inference provider listed on OpenRouter. Its prices here are the ones OpenRouter’s public API reported for its endpoints.",["Amazon Bedrock"],{"$k":["studySlug","chartId","series","point","metric","label","value","unit","display","context"],"$r":[["inference-provider-index","provider-prices-claude-haiku-4-5","Input","Amazon Bedrock","provider-prices-claude-haiku-4-5/Input","Claude Haiku 4.5: price per million tokens by provider (Input)",1,"usd","$1.00","Claude Haiku 4.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-claude-haiku-4-5","Output","Amazon Bedrock","provider-prices-claude-haiku-4-5/Output","Claude Haiku 4.5: price per million tokens by provider (Output)",5,"usd","$5.00","Claude Haiku 4.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-claude-haiku-4-5","Cache read","Amazon Bedrock","provider-prices-claude-haiku-4-5/Cache read","Claude Haiku 4.5: price per million tokens by provider (Cache read)",0.1,"usd","$0.10","Claude Haiku 4.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-claude-sonnet-5","Input","Amazon Bedrock","provider-prices-claude-sonnet-5/Input","Claude Sonnet 5: price per million tokens by provider (Input)",2,"usd","$2.00","Claude Sonnet 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-claude-sonnet-5","Output","Amazon Bedrock","provider-prices-claude-sonnet-5/Output","Claude Sonnet 5: price per million tokens by provider (Output)",10,"usd","$10.00","Claude Sonnet 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-claude-sonnet-5","Cache read","Amazon Bedrock","provider-prices-claude-sonnet-5/Cache read","Claude Sonnet 5: price per million tokens by provider (Cache read)",0.2,"usd","$0.20","Claude Sonnet 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-claude-sonnet-5-5","Input","Amazon Bedrock","provider-prices-claude-sonnet-5-5/Input","Claude Sonnet 5.5: price per million tokens by provider (Input)",2,"usd","$2.00","Claude Sonnet 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-claude-sonnet-5-5","Output","Amazon Bedrock","provider-prices-claude-sonnet-5-5/Output","Claude Sonnet 5.5: price per million tokens by provider (Output)",10,"usd","$10.00","Claude Sonnet 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-claude-sonnet-5-5","Cache read","Amazon Bedrock","provider-prices-claude-sonnet-5-5/Cache read","Claude Sonnet 5.5: price per million tokens by provider (Cache read)",0.2,"usd","$0.20","Claude Sonnet 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-claude-opus-4-8","Input","Amazon Bedrock","provider-prices-claude-opus-4-8/Input","Claude Opus 4.8: price per million tokens by provider (Input)",5,"usd","$5.00","Claude Opus 4.8 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-claude-opus-4-8","Output","Amazon Bedrock","provider-prices-claude-opus-4-8/Output","Claude Opus 4.8: price per million tokens by provider (Output)",25,"usd","$25.00","Claude Opus 4.8 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-claude-opus-4-8","Cache read","Amazon Bedrock","provider-prices-claude-opus-4-8/Cache read","Claude Opus 4.8: price per million tokens by provider (Cache read)",0.5,"usd","$0.50","Claude Opus 4.8 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-claude-opus-5","Input","Amazon Bedrock","provider-prices-claude-opus-5/Input","Claude Opus 5: price per million tokens by provider (Input)",5,"usd","$5.00","Claude Opus 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-claude-opus-5","Output","Amazon Bedrock","provider-prices-claude-opus-5/Output","Claude Opus 5: price per million tokens by provider (Output)",25,"usd","$25.00","Claude Opus 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-claude-opus-5","Cache read","Amazon Bedrock","provider-prices-claude-opus-5/Cache read","Claude Opus 5: price per million tokens by provider (Cache read)",0.5,"usd","$0.50","Claude Opus 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-claude-opus-5-5","Input","Amazon Bedrock","provider-prices-claude-opus-5-5/Input","Claude Opus 5.5: price per million tokens by provider (Input)",4,"usd","$4.00","Claude Opus 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-claude-opus-5-5","Output","Amazon Bedrock","provider-prices-claude-opus-5-5/Output","Claude Opus 5.5: price per million tokens by provider (Output)",20,"usd","$20.00","Claude Opus 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-claude-opus-5-5","Cache read","Amazon Bedrock","provider-prices-claude-opus-5-5/Cache read","Claude Opus 5.5: price per million tokens by provider (Cache read)",0.2,"usd","$0.20","Claude Opus 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-claude-fable-5-1","Input","Amazon Bedrock","provider-prices-claude-fable-5-1/Input","Claude Fable 5.1: price per million tokens by provider (Input)",10,"usd","$10.00","Claude Fable 5.1 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-claude-fable-5-1","Output","Amazon Bedrock","provider-prices-claude-fable-5-1/Output","Claude Fable 5.1: price per million tokens by provider (Output)",50,"usd","$50.00","Claude Fable 5.1 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-claude-fable-5-1","Cache read","Amazon Bedrock","provider-prices-claude-fable-5-1/Cache read","Claude Fable 5.1: price per million tokens by provider (Cache read)",0.25,"usd","$0.25","Claude Fable 5.1 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-gpt-oss-120b","Input","Amazon Bedrock","provider-prices-gpt-oss-120b/Input","gpt-oss-120b: price per million tokens by provider (Input)",0.15,"usd","$0.15","gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-gpt-oss-120b","Output","Amazon Bedrock","provider-prices-gpt-oss-120b/Output","gpt-oss-120b: price per million tokens by provider (Output)",0.6,"usd","$0.60","gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"]]}],["azure","Azure","Microsoft","provider","Microsoft’s cloud model platform. An inference provider listed on OpenRouter. Its prices here are the ones OpenRouter’s public API reported for its endpoints.",["Azure"],{"$k":["studySlug","chartId","series","point","metric","label","value","unit","display","context"],"$r":[["inference-provider-index","provider-prices-claude-haiku-4-5","Input","Azure","provider-prices-claude-haiku-4-5/Input","Claude Haiku 4.5: price per million tokens by provider (Input)",1,"usd","$1.00","Claude Haiku 4.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-claude-haiku-4-5","Output","Azure","provider-prices-claude-haiku-4-5/Output","Claude Haiku 4.5: price per million tokens by provider (Output)",5,"usd","$5.00","Claude Haiku 4.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-claude-haiku-4-5","Cache read","Azure","provider-prices-claude-haiku-4-5/Cache read","Claude Haiku 4.5: price per million tokens by provider (Cache read)",0.1,"usd","$0.10","Claude Haiku 4.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-claude-sonnet-5","Input","Azure","provider-prices-claude-sonnet-5/Input","Claude Sonnet 5: price per million tokens by provider (Input)",2,"usd","$2.00","Claude Sonnet 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-claude-sonnet-5","Output","Azure","provider-prices-claude-sonnet-5/Output","Claude Sonnet 5: price per million tokens by provider (Output)",10,"usd","$10.00","Claude Sonnet 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-claude-sonnet-5","Cache read","Azure","provider-prices-claude-sonnet-5/Cache read","Claude Sonnet 5: price per million tokens by provider (Cache read)",0.2,"usd","$0.20","Claude Sonnet 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-claude-sonnet-5-5","Input","Azure","provider-prices-claude-sonnet-5-5/Input","Claude Sonnet 5.5: price per million tokens by provider (Input)",2,"usd","$2.00","Claude Sonnet 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-claude-sonnet-5-5","Output","Azure","provider-prices-claude-sonnet-5-5/Output","Claude Sonnet 5.5: price per million tokens by provider (Output)",10,"usd","$10.00","Claude Sonnet 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-claude-sonnet-5-5","Cache read","Azure","provider-prices-claude-sonnet-5-5/Cache read","Claude Sonnet 5.5: price per million tokens by provider (Cache read)",0.2,"usd","$0.20","Claude Sonnet 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-claude-opus-4-8","Input","Azure","provider-prices-claude-opus-4-8/Input","Claude Opus 4.8: price per million tokens by provider (Input)",5,"usd","$5.00","Claude Opus 4.8 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-claude-opus-4-8","Output","Azure","provider-prices-claude-opus-4-8/Output","Claude Opus 4.8: price per million tokens by provider (Output)",25,"usd","$25.00","Claude Opus 4.8 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-claude-opus-4-8","Cache read","Azure","provider-prices-claude-opus-4-8/Cache read","Claude Opus 4.8: price per million tokens by provider (Cache read)",0.5,"usd","$0.50","Claude Opus 4.8 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-claude-opus-5","Input","Azure","provider-prices-claude-opus-5/Input","Claude Opus 5: price per million tokens by provider (Input)",5,"usd","$5.00","Claude Opus 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-claude-opus-5","Output","Azure","provider-prices-claude-opus-5/Output","Claude Opus 5: price per million tokens by provider (Output)",25,"usd","$25.00","Claude Opus 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-claude-opus-5","Cache read","Azure","provider-prices-claude-opus-5/Cache read","Claude Opus 5: price per million tokens by provider (Cache read)",0.5,"usd","$0.50","Claude Opus 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-claude-opus-5-5","Input","Azure","provider-prices-claude-opus-5-5/Input","Claude Opus 5.5: price per million tokens by provider (Input)",4,"usd","$4.00","Claude Opus 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-claude-opus-5-5","Output","Azure","provider-prices-claude-opus-5-5/Output","Claude Opus 5.5: price per million tokens by provider (Output)",20,"usd","$20.00","Claude Opus 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-claude-opus-5-5","Cache read","Azure","provider-prices-claude-opus-5-5/Cache read","Claude Opus 5.5: price per million tokens by provider (Cache read)",0.2,"usd","$0.20","Claude Opus 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-claude-fable-5-1","Input","Azure","provider-prices-claude-fable-5-1/Input","Claude Fable 5.1: price per million tokens by provider (Input)",10,"usd","$10.00","Claude Fable 5.1 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-claude-fable-5-1","Output","Azure","provider-prices-claude-fable-5-1/Output","Claude Fable 5.1: price per million tokens by provider (Output)",50,"usd","$50.00","Claude Fable 5.1 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-claude-fable-5-1","Cache read","Azure","provider-prices-claude-fable-5-1/Cache read","Claude Fable 5.1: price per million tokens by provider (Cache read)",0.25,"usd","$0.25","Claude Fable 5.1 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-gpt-6-sol","Input","Azure","provider-prices-gpt-6-sol/Input","GPT-6 Sol: price per million tokens by provider (Input)",2,"usd","$2.00","GPT-6 Sol · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-gpt-6-sol","Output","Azure","provider-prices-gpt-6-sol/Output","GPT-6 Sol: price per million tokens by provider (Output)",10,"usd","$10.00","GPT-6 Sol · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-gpt-6-sol","Cache read","Azure","provider-prices-gpt-6-sol/Cache read","GPT-6 Sol: price per million tokens by provider (Cache read)",0.2,"usd","$0.20","GPT-6 Sol · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-gpt-6-luna","Input","Azure","provider-prices-gpt-6-luna/Input","GPT-6 Luna: price per million tokens by provider (Input)",0.1,"usd","$0.10","GPT-6 Luna · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-gpt-6-luna","Output","Azure","provider-prices-gpt-6-luna/Output","GPT-6 Luna: price per million tokens by provider (Output)",0.5,"usd","$0.50","GPT-6 Luna · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-gpt-6-luna","Cache read","Azure","provider-prices-gpt-6-luna/Cache read","GPT-6 Luna: price per million tokens by provider (Cache read)",0.01,"usd","$0.010","GPT-6 Luna · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-gpt-6-astra","Input","Azure","provider-prices-gpt-6-astra/Input","GPT-6 Astra: price per million tokens by provider (Input)",10,"usd","$10.00","GPT-6 Astra · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-gpt-6-astra","Output","Azure","provider-prices-gpt-6-astra/Output","GPT-6 Astra: price per million tokens by provider (Output)",50,"usd","$50.00","GPT-6 Astra · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-gpt-6-astra","Cache read","Azure","provider-prices-gpt-6-astra/Cache read","GPT-6 Astra: price per million tokens by provider (Cache read)",1,"usd","$1.00","GPT-6 Astra · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-gpt-5-5","Input","Azure","provider-prices-gpt-5-5/Input","GPT-5.5: price per million tokens by provider (Input)",5,"usd","$5.00","GPT-5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-gpt-5-5","Output","Azure","provider-prices-gpt-5-5/Output","GPT-5.5: price per million tokens by provider (Output)",30,"usd","$30.00","GPT-5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-gpt-5-5","Cache read","Azure","provider-prices-gpt-5-5/Cache read","GPT-5.5: price per million tokens by provider (Cache read)",0.5,"usd","$0.50","GPT-5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"]]}],["claude-platform-on-aws","Claude Platform on AWS","Anthropic","provider","Anthropic’s Claude platform hosted on AWS. An inference provider listed on OpenRouter. Its prices here are the ones OpenRouter’s public API reported for its endpoints.",["Claude Platform on AWS"],{"$k":["studySlug","chartId","series","point","metric","label","value","unit","display","context"],"$r":[["inference-provider-index","provider-prices-claude-sonnet-5","Input","Claude Platform on AWS","provider-prices-claude-sonnet-5/Input","Claude Sonnet 5: price per million tokens by provider (Input)",2,"usd","$2.00","Claude Sonnet 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-claude-sonnet-5","Output","Claude Platform on AWS","provider-prices-claude-sonnet-5/Output","Claude Sonnet 5: price per million tokens by provider (Output)",10,"usd","$10.00","Claude Sonnet 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-claude-sonnet-5","Cache read","Claude Platform on AWS","provider-prices-claude-sonnet-5/Cache read","Claude Sonnet 5: price per million tokens by provider (Cache read)",0.2,"usd","$0.20","Claude Sonnet 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-claude-sonnet-5-5","Input","Claude Platform on AWS","provider-prices-claude-sonnet-5-5/Input","Claude Sonnet 5.5: price per million tokens by provider (Input)",2,"usd","$2.00","Claude Sonnet 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-claude-sonnet-5-5","Output","Claude Platform on AWS","provider-prices-claude-sonnet-5-5/Output","Claude Sonnet 5.5: price per million tokens by provider (Output)",10,"usd","$10.00","Claude Sonnet 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-claude-sonnet-5-5","Cache read","Claude Platform on AWS","provider-prices-claude-sonnet-5-5/Cache read","Claude Sonnet 5.5: price per million tokens by provider (Cache read)",0.2,"usd","$0.20","Claude Sonnet 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-claude-opus-4-8","Input","Claude Platform on AWS","provider-prices-claude-opus-4-8/Input","Claude Opus 4.8: price per million tokens by provider (Input)",5,"usd","$5.00","Claude Opus 4.8 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-claude-opus-4-8","Output","Claude Platform on AWS","provider-prices-claude-opus-4-8/Output","Claude Opus 4.8: price per million tokens by provider (Output)",25,"usd","$25.00","Claude Opus 4.8 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-claude-opus-4-8","Cache read","Claude Platform on AWS","provider-prices-claude-opus-4-8/Cache read","Claude Opus 4.8: price per million tokens by provider (Cache read)",0.5,"usd","$0.50","Claude Opus 4.8 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-claude-opus-5","Input","Claude Platform on AWS","provider-prices-claude-opus-5/Input","Claude Opus 5: price per million tokens by provider (Input)",5,"usd","$5.00","Claude Opus 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-claude-opus-5","Output","Claude Platform on AWS","provider-prices-claude-opus-5/Output","Claude Opus 5: price per million tokens by provider (Output)",25,"usd","$25.00","Claude Opus 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-claude-opus-5","Cache read","Claude Platform on AWS","provider-prices-claude-opus-5/Cache read","Claude Opus 5: price per million tokens by provider (Cache read)",0.5,"usd","$0.50","Claude Opus 5 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-claude-opus-5-5","Input","Claude Platform on AWS","provider-prices-claude-opus-5-5/Input","Claude Opus 5.5: price per million tokens by provider (Input)",4,"usd","$4.00","Claude Opus 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-claude-opus-5-5","Output","Claude Platform on AWS","provider-prices-claude-opus-5-5/Output","Claude Opus 5.5: price per million tokens by provider (Output)",20,"usd","$20.00","Claude Opus 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-claude-opus-5-5","Cache read","Claude Platform on AWS","provider-prices-claude-opus-5-5/Cache read","Claude Opus 5.5: price per million tokens by provider (Cache read)",0.2,"usd","$0.20","Claude Opus 5.5 · reported by OpenRouter’s public API, snapshot 2026-10-06"]]}],["groq","Groq","Groq","provider","An inference provider listed on OpenRouter. Its prices here are the ones OpenRouter’s public API reported for its endpoints.",["Groq"],{"$k":["studySlug","chartId","series","point","metric","label","value","unit","display","context"],"$r":[["inference-provider-index","provider-prices-gpt-oss-120b","Input","Groq","provider-prices-gpt-oss-120b/Input","gpt-oss-120b: price per million tokens by provider (Input)",0.15,"usd","$0.15","gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-gpt-oss-120b","Output","Groq","provider-prices-gpt-oss-120b/Output","gpt-oss-120b: price per million tokens by provider (Output)",0.6,"usd","$0.60","gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-gpt-oss-120b","Cache read","Groq","provider-prices-gpt-oss-120b/Cache read","gpt-oss-120b: price per million tokens by provider (Cache read)",0.075,"usd","$0.075","gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-llama-3-3-70b-instruct","Input","Groq","provider-prices-llama-3-3-70b-instruct/Input","Llama 3.3 70B Instruct: price per million tokens by provider (Input)",0.59,"usd","$0.59","Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-llama-3-3-70b-instruct","Output","Groq","provider-prices-llama-3-3-70b-instruct/Output","Llama 3.3 70B Instruct: price per million tokens by provider (Output)",0.79,"usd","$0.79","Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-llama-3-3-70b-instruct","Cache read","Groq","provider-prices-llama-3-3-70b-instruct/Cache read","Llama 3.3 70B Instruct: price per million tokens by provider (Cache read)",0.295,"usd","$0.29","Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06"]]}],["together","Together AI","Together AI","provider","An inference provider listed on OpenRouter. Its prices here are the ones OpenRouter’s public API reported for its endpoints.",["Together"],{"$k":["studySlug","chartId","series","point","metric","label","value","unit","display","context"],"$r":[["inference-provider-index","provider-prices-gpt-oss-120b","Input","Together","provider-prices-gpt-oss-120b/Input","gpt-oss-120b: price per million tokens by provider (Input)",0.15,"usd","$0.15","gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-gpt-oss-120b","Output","Together","provider-prices-gpt-oss-120b/Output","gpt-oss-120b: price per million tokens by provider (Output)",0.6,"usd","$0.60","gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-llama-3-3-70b-instruct","Input","Together","provider-prices-llama-3-3-70b-instruct/Input","Llama 3.3 70B Instruct: price per million tokens by provider (Input)",1.04,"usd","$1.04","Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-llama-3-3-70b-instruct","Output","Together","provider-prices-llama-3-3-70b-instruct/Output","Llama 3.3 70B Instruct: price per million tokens by provider (Output)",1.04,"usd","$1.04","Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-kimi-k3","Input","Together","provider-prices-kimi-k3/Input","Kimi K3: price per million tokens by provider (Input)",2.7,"usd","$2.70","Kimi K3 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-kimi-k3","Output","Together","provider-prices-kimi-k3/Output","Kimi K3: price per million tokens by provider (Output)",13.5,"usd","$13.50","Kimi K3 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-kimi-k3","Cache read","Together","provider-prices-kimi-k3/Cache read","Kimi K3: price per million tokens by provider (Cache read)",0.27,"usd","$0.27","Kimi K3 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-glm-5-3","Input","Together","provider-prices-glm-5-3/Input","GLM 5.3: price per million tokens by provider (Input)",1.4,"usd","$1.40","GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-glm-5-3","Output","Together","provider-prices-glm-5-3/Output","GLM 5.3: price per million tokens by provider (Output)",4.4,"usd","$4.40","GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-glm-5-3","Cache read","Together","provider-prices-glm-5-3/Cache read","GLM 5.3: price per million tokens by provider (Cache read)",0.26,"usd","$0.26","GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"]]}],["fireworks","Fireworks AI","Fireworks AI","provider","An inference provider listed on OpenRouter. Its prices here are the ones OpenRouter’s public API reported for its endpoints.",["Fireworks"],{"$k":["studySlug","chartId","series","point","metric","label","value","unit","display","context"],"$r":[["inference-provider-index","provider-prices-kimi-k3","Input","Fireworks","provider-prices-kimi-k3/Input","Kimi K3: price per million tokens by provider (Input)",3,"usd","$3.00","Kimi K3 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-kimi-k3","Output","Fireworks","provider-prices-kimi-k3/Output","Kimi K3: price per million tokens by provider (Output)",15,"usd","$15.00","Kimi K3 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-kimi-k3","Cache read","Fireworks","provider-prices-kimi-k3/Cache read","Kimi K3: price per million tokens by provider (Cache read)",0.3,"usd","$0.30","Kimi K3 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-glm-5-3","Input","Fireworks","provider-prices-glm-5-3/Input","GLM 5.3: price per million tokens by provider (Input)",1.4,"usd","$1.40","GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-glm-5-3","Output","Fireworks","provider-prices-glm-5-3/Output","GLM 5.3: price per million tokens by provider (Output)",4.4,"usd","$4.40","GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-glm-5-3","Cache read","Fireworks","provider-prices-glm-5-3/Cache read","GLM 5.3: price per million tokens by provider (Cache read)",0.26,"usd","$0.26","GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"]]}],["deepinfra","DeepInfra","DeepInfra","provider","An inference provider listed on OpenRouter. Its prices here are the ones OpenRouter’s public API reported for its endpoints.",["DeepInfra"],{"$k":["studySlug","chartId","series","point","metric","label","value","unit","display","context"],"$r":[["inference-provider-index","provider-prices-gpt-oss-120b","Input","DeepInfra (bf16)","provider-prices-gpt-oss-120b/Input","gpt-oss-120b: price per million tokens by provider (Input)",0.037,"usd","$0.037","bf16 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-gpt-oss-120b","Output","DeepInfra (bf16)","provider-prices-gpt-oss-120b/Output","gpt-oss-120b: price per million tokens by provider (Output)",0.17,"usd","$0.17","bf16 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-llama-3-3-70b-instruct","Input","DeepInfra (fp8)","provider-prices-llama-3-3-70b-instruct/Input","Llama 3.3 70B Instruct: price per million tokens by provider (Input)",0.1,"usd","$0.10","fp8 · Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-llama-3-3-70b-instruct","Output","DeepInfra (fp8)","provider-prices-llama-3-3-70b-instruct/Output","Llama 3.3 70B Instruct: price per million tokens by provider (Output)",0.32,"usd","$0.32","fp8 · Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-deepseek-v4-pro","Input","DeepInfra (fp8)","provider-prices-deepseek-v4-pro/Input","DeepSeek V4 Pro 0423: price per million tokens by provider (Input)",1.3,"usd","$1.30","fp8 · DeepSeek V4 Pro 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-deepseek-v4-pro","Output","DeepInfra (fp8)","provider-prices-deepseek-v4-pro/Output","DeepSeek V4 Pro 0423: price per million tokens by provider (Output)",2.6,"usd","$2.60","fp8 · DeepSeek V4 Pro 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-deepseek-v4-pro","Cache read","DeepInfra (fp8)","provider-prices-deepseek-v4-pro/Cache read","DeepSeek V4 Pro 0423: price per million tokens by provider (Cache read)",0.1,"usd","$0.10","fp8 · DeepSeek V4 Pro 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-deepseek-v4-flash","Input","DeepInfra (fp8)","provider-prices-deepseek-v4-flash/Input","DeepSeek V4 Flash 0423: price per million tokens by provider (Input)",0.09,"usd","$0.090","fp8 · DeepSeek V4 Flash 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-deepseek-v4-flash","Output","DeepInfra (fp8)","provider-prices-deepseek-v4-flash/Output","DeepSeek V4 Flash 0423: price per million tokens by provider (Output)",0.18,"usd","$0.18","fp8 · DeepSeek V4 Flash 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-deepseek-v4-flash","Cache read","DeepInfra (fp8)","provider-prices-deepseek-v4-flash/Cache read","DeepSeek V4 Flash 0423: price per million tokens by provider (Cache read)",0.018,"usd","$0.018","fp8 · DeepSeek V4 Flash 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-kimi-k3","Input","DeepInfra (mxfp4)","provider-prices-kimi-k3/Input","Kimi K3: price per million tokens by provider (Input)",2.85,"usd","$2.85","mxfp4 · Kimi K3 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-kimi-k3","Output","DeepInfra (mxfp4)","provider-prices-kimi-k3/Output","Kimi K3: price per million tokens by provider (Output)",14.25,"usd","$14.25","mxfp4 · Kimi K3 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-kimi-k3","Cache read","DeepInfra (mxfp4)","provider-prices-kimi-k3/Cache read","Kimi K3: price per million tokens by provider (Cache read)",0.285,"usd","$0.28","mxfp4 · Kimi K3 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-glm-5-3","Input","DeepInfra (fp4)","provider-prices-glm-5-3/Input","GLM 5.3: price per million tokens by provider (Input)",0.5625,"usd","$0.56","fp4 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-glm-5-3","Output","DeepInfra (fp4)","provider-prices-glm-5-3/Output","GLM 5.3: price per million tokens by provider (Output)",2.5,"usd","$2.50","fp4 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-glm-5-3","Cache read","DeepInfra (fp4)","provider-prices-glm-5-3/Cache read","GLM 5.3: price per million tokens by provider (Cache read)",0.125,"usd","$0.13","fp4 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"]]}],["cerebras","Cerebras","Cerebras","provider","An inference provider listed on OpenRouter. Its prices here are the ones OpenRouter’s public API reported for its endpoints.",["Cerebras"],{"$k":["studySlug","chartId","series","point","metric","label","value","unit","display","context"],"$r":[["inference-provider-index","provider-prices-gpt-oss-120b","Input","Cerebras (fp16)","provider-prices-gpt-oss-120b/Input","gpt-oss-120b: price per million tokens by provider (Input)",0.35,"usd","$0.35","fp16 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-gpt-oss-120b","Output","Cerebras (fp16)","provider-prices-gpt-oss-120b/Output","gpt-oss-120b: price per million tokens by provider (Output)",0.75,"usd","$0.75","fp16 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-gpt-oss-120b","Cache read","Cerebras (fp16)","provider-prices-gpt-oss-120b/Cache read","gpt-oss-120b: price per million tokens by provider (Cache read)",0.35,"usd","$0.35","fp16 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"]]}],["sambanova","SambaNova","SambaNova","provider","An inference provider listed on OpenRouter. Its prices here are the ones OpenRouter’s public API reported for its endpoints.",["SambaNova"],{"$k":["studySlug","chartId","series","point","metric","label","value","unit","display","context"],"$r":[["inference-provider-index","provider-prices-gpt-oss-120b","Input","SambaNova","provider-prices-gpt-oss-120b/Input","gpt-oss-120b: price per million tokens by provider (Input)",0.14,"usd","$0.14","gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-gpt-oss-120b","Output","SambaNova","provider-prices-gpt-oss-120b/Output","gpt-oss-120b: price per million tokens by provider (Output)",0.95,"usd","$0.95","gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-llama-3-3-70b-instruct","Input","SambaNova","provider-prices-llama-3-3-70b-instruct/Input","Llama 3.3 70B Instruct: price per million tokens by provider (Input)",0.45,"usd","$0.45","Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-llama-3-3-70b-instruct","Output","SambaNova","provider-prices-llama-3-3-70b-instruct/Output","Llama 3.3 70B Instruct: price per million tokens by provider (Output)",0.9,"usd","$0.90","Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06"]]}],["nebius","Nebius","Nebius","provider","An inference provider listed on OpenRouter. Its prices here are the ones OpenRouter’s public API reported for its endpoints.",["Nebius"],{"$k":["studySlug","chartId","series","point","metric","label","value","unit","display","context"],"$r":[["inference-provider-index","provider-prices-gpt-oss-120b","Input","Nebius (fp4)","provider-prices-gpt-oss-120b/Input","gpt-oss-120b: price per million tokens by provider (Input)",0.15,"usd","$0.15","fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-gpt-oss-120b","Output","Nebius (fp4)","provider-prices-gpt-oss-120b/Output","gpt-oss-120b: price per million tokens by provider (Output)",0.6,"usd","$0.60","fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-glm-5-3","Input","Nebius (fp4)","provider-prices-glm-5-3/Input","GLM 5.3: price per million tokens by provider (Input)",1.4,"usd","$1.40","fp4 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-glm-5-3","Output","Nebius (fp4)","provider-prices-glm-5-3/Output","GLM 5.3: price per million tokens by provider (Output)",4.4,"usd","$4.40","fp4 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"]]}],["parasail","Parasail","Parasail","provider","An inference provider listed on OpenRouter. Its prices here are the ones OpenRouter’s public API reported for its endpoints.",["Parasail"],{"$k":["studySlug","chartId","series","point","metric","label","value","unit","display","context"],"$r":[["inference-provider-index","provider-prices-gpt-oss-120b","Input","Parasail (fp4)","provider-prices-gpt-oss-120b/Input","gpt-oss-120b: price per million tokens by provider (Input)",0.1,"usd","$0.10","fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-gpt-oss-120b","Output","Parasail (fp4)","provider-prices-gpt-oss-120b/Output","gpt-oss-120b: price per million tokens by provider (Output)",0.75,"usd","$0.75","fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-gpt-oss-120b","Cache read","Parasail (fp4)","provider-prices-gpt-oss-120b/Cache read","gpt-oss-120b: price per million tokens by provider (Cache read)",0.055,"usd","$0.055","fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-llama-4-maverick","Input","Parasail (fp8)","provider-prices-llama-4-maverick/Input","Llama 4 Maverick: price per million tokens by provider (Input)",0.35,"usd","$0.35","fp8 · Llama 4 Maverick · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-llama-4-maverick","Output","Parasail (fp8)","provider-prices-llama-4-maverick/Output","Llama 4 Maverick: price per million tokens by provider (Output)",1,"usd","$1.00","fp8 · Llama 4 Maverick · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-llama-3-3-70b-instruct","Input","Parasail (fp8)","provider-prices-llama-3-3-70b-instruct/Input","Llama 3.3 70B Instruct: price per million tokens by provider (Input)",0.22,"usd","$0.22","fp8 · Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-llama-3-3-70b-instruct","Output","Parasail (fp8)","provider-prices-llama-3-3-70b-instruct/Output","Llama 3.3 70B Instruct: price per million tokens by provider (Output)",0.5,"usd","$0.50","fp8 · Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-llama-3-3-70b-instruct","Cache read","Parasail (fp8)","provider-prices-llama-3-3-70b-instruct/Cache read","Llama 3.3 70B Instruct: price per million tokens by provider (Cache read)",0.11,"usd","$0.11","fp8 · Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-deepseek-v4-pro","Input","Parasail (fp8)","provider-prices-deepseek-v4-pro/Input","DeepSeek V4 Pro 0423: price per million tokens by provider (Input)",0.45,"usd","$0.45","fp8 · DeepSeek V4 Pro 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-deepseek-v4-pro","Output","Parasail (fp8)","provider-prices-deepseek-v4-pro/Output","DeepSeek V4 Pro 0423: price per million tokens by provider (Output)",3.48,"usd","$3.48","fp8 · DeepSeek V4 Pro 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-deepseek-v4-pro","Cache read","Parasail (fp8)","provider-prices-deepseek-v4-pro/Cache read","DeepSeek V4 Pro 0423: price per million tokens by provider (Cache read)",0.1,"usd","$0.10","fp8 · DeepSeek V4 Pro 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-deepseek-v4-flash","Input","Parasail (fp8)","provider-prices-deepseek-v4-flash/Input","DeepSeek V4 Flash 0423: price per million tokens by provider (Input)",0.14,"usd","$0.14","fp8 · DeepSeek V4 Flash 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-deepseek-v4-flash","Output","Parasail (fp8)","provider-prices-deepseek-v4-flash/Output","DeepSeek V4 Flash 0423: price per million tokens by provider (Output)",0.28,"usd","$0.28","fp8 · DeepSeek V4 Flash 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-deepseek-v4-flash","Cache read","Parasail (fp8)","provider-prices-deepseek-v4-flash/Cache read","DeepSeek V4 Flash 0423: price per million tokens by provider (Cache read)",0.07,"usd","$0.070","fp8 · DeepSeek V4 Flash 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-kimi-k3","Input","Parasail (fp4)","provider-prices-kimi-k3/Input","Kimi K3: price per million tokens by provider (Input)",3,"usd","$3.00","fp4 · Kimi K3 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-kimi-k3","Output","Parasail (fp4)","provider-prices-kimi-k3/Output","Kimi K3: price per million tokens by provider (Output)",15,"usd","$15.00","fp4 · Kimi K3 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-kimi-k3","Cache read","Parasail (fp4)","provider-prices-kimi-k3/Cache read","Kimi K3: price per million tokens by provider (Cache read)",0.3,"usd","$0.30","fp4 · Kimi K3 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-glm-5-3","Input","Parasail (fp8)","provider-prices-glm-5-3/Input","GLM 5.3: price per million tokens by provider (Input)",1.4,"usd","$1.40","fp8 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-glm-5-3","Output","Parasail (fp8)","provider-prices-glm-5-3/Output","GLM 5.3: price per million tokens by provider (Output)",4.4,"usd","$4.40","fp8 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-glm-5-3","Cache read","Parasail (fp8)","provider-prices-glm-5-3/Cache read","GLM 5.3: price per million tokens by provider (Cache read)",0.26,"usd","$0.26","fp8 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"]]}],["novita","Novita AI","Novita AI","provider","An inference provider listed on OpenRouter. Its prices here are the ones OpenRouter’s public API reported for its endpoints.",["Novita"],{"$k":["studySlug","chartId","series","point","metric","label","value","unit","display","context"],"$r":[["inference-provider-index","provider-prices-gpt-oss-120b","Input","Novita (fp4)","provider-prices-gpt-oss-120b/Input","gpt-oss-120b: price per million tokens by provider (Input)",0.05,"usd","$0.050","fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-gpt-oss-120b","Output","Novita (fp4)","provider-prices-gpt-oss-120b/Output","gpt-oss-120b: price per million tokens by provider (Output)",0.25,"usd","$0.25","fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-llama-4-maverick","Input","Novita (fp8)","provider-prices-llama-4-maverick/Input","Llama 4 Maverick: price per million tokens by provider (Input)",0.27,"usd","$0.27","fp8 · Llama 4 Maverick · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-llama-4-maverick","Output","Novita (fp8)","provider-prices-llama-4-maverick/Output","Llama 4 Maverick: price per million tokens by provider (Output)",0.85,"usd","$0.85","fp8 · Llama 4 Maverick · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-llama-3-3-70b-instruct","Input","Novita (bf16)","provider-prices-llama-3-3-70b-instruct/Input","Llama 3.3 70B Instruct: price per million tokens by provider (Input)",0.135,"usd","$0.14","bf16 · Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-llama-3-3-70b-instruct","Output","Novita (bf16)","provider-prices-llama-3-3-70b-instruct/Output","Llama 3.3 70B Instruct: price per million tokens by provider (Output)",0.4,"usd","$0.40","bf16 · Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-deepseek-v4-pro","Input","Novita (fp8)","provider-prices-deepseek-v4-pro/Input","DeepSeek V4 Pro 0423: price per million tokens by provider (Input)",1.6,"usd","$1.60","fp8 · DeepSeek V4 Pro 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-deepseek-v4-pro","Output","Novita (fp8)","provider-prices-deepseek-v4-pro/Output","DeepSeek V4 Pro 0423: price per million tokens by provider (Output)",3.2,"usd","$3.20","fp8 · DeepSeek V4 Pro 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-deepseek-v4-pro","Cache read","Novita (fp8)","provider-prices-deepseek-v4-pro/Cache read","DeepSeek V4 Pro 0423: price per million tokens by provider (Cache read)",0.135,"usd","$0.14","fp8 · DeepSeek V4 Pro 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-deepseek-v4-flash","Input","Novita (fp8)","provider-prices-deepseek-v4-flash/Input","DeepSeek V4 Flash 0423: price per million tokens by provider (Input)",0.14,"usd","$0.14","fp8 · DeepSeek V4 Flash 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-deepseek-v4-flash","Output","Novita (fp8)","provider-prices-deepseek-v4-flash/Output","DeepSeek V4 Flash 0423: price per million tokens by provider (Output)",0.28,"usd","$0.28","fp8 · DeepSeek V4 Flash 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-deepseek-v4-flash","Cache read","Novita (fp8)","provider-prices-deepseek-v4-flash/Cache read","DeepSeek V4 Flash 0423: price per million tokens by provider (Cache read)",0.028,"usd","$0.028","fp8 · DeepSeek V4 Flash 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-glm-5-3","Input","Novita (fp8)","provider-prices-glm-5-3/Input","GLM 5.3: price per million tokens by provider (Input)",0.42,"usd","$0.42","fp8 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-glm-5-3","Output","Novita (fp8)","provider-prices-glm-5-3/Output","GLM 5.3: price per million tokens by provider (Output)",1.32,"usd","$1.32","fp8 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-glm-5-3","Cache read","Novita (fp8)","provider-prices-glm-5-3/Cache read","GLM 5.3: price per million tokens by provider (Cache read)",0.078,"usd","$0.078","fp8 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"]]}],["baseten","Baseten","Baseten","provider","An inference provider listed on OpenRouter. Its prices here are the ones OpenRouter’s public API reported for its endpoints.",["BaseTen"],{"$k":["studySlug","chartId","series","point","metric","label","value","unit","display","context"],"$r":[["inference-provider-index","provider-prices-gpt-oss-120b","Input","BaseTen (fp4)","provider-prices-gpt-oss-120b/Input","gpt-oss-120b: price per million tokens by provider (Input)",0.1,"usd","$0.10","fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-gpt-oss-120b","Output","BaseTen (fp4)","provider-prices-gpt-oss-120b/Output","gpt-oss-120b: price per million tokens by provider (Output)",0.5,"usd","$0.50","fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-gpt-oss-120b","Cache read","BaseTen (fp4)","provider-prices-gpt-oss-120b/Cache read","gpt-oss-120b: price per million tokens by provider (Cache read)",0.1,"usd","$0.10","fp4 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-kimi-k3","Input","BaseTen (fp8)","provider-prices-kimi-k3/Input","Kimi K3: price per million tokens by provider (Input)",3,"usd","$3.00","fp8 · Kimi K3 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-kimi-k3","Output","BaseTen (fp8)","provider-prices-kimi-k3/Output","Kimi K3: price per million tokens by provider (Output)",15,"usd","$15.00","fp8 · Kimi K3 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-kimi-k3","Cache read","BaseTen (fp8)","provider-prices-kimi-k3/Cache read","Kimi K3: price per million tokens by provider (Cache read)",0.3,"usd","$0.30","fp8 · Kimi K3 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-glm-5-3","Input","BaseTen (fp4)","provider-prices-glm-5-3/Input","GLM 5.3: price per million tokens by provider (Input)",1.4,"usd","$1.40","fp4 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-glm-5-3","Output","BaseTen (fp4)","provider-prices-glm-5-3/Output","GLM 5.3: price per million tokens by provider (Output)",4.4,"usd","$4.40","fp4 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-glm-5-3","Cache read","BaseTen (fp4)","provider-prices-glm-5-3/Cache read","GLM 5.3: price per million tokens by provider (Cache read)",0.14,"usd","$0.14","fp4 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"]]}],["cloudflare","Cloudflare Workers AI","Cloudflare","provider","An inference provider listed on OpenRouter. Its prices here are the ones OpenRouter’s public API reported for its endpoints.",["Cloudflare"],{"$k":["studySlug","chartId","series","point","metric","label","value","unit","display","context"],"$r":[["inference-provider-index","provider-prices-llama-3-3-70b-instruct","Input","Cloudflare (fp8)","provider-prices-llama-3-3-70b-instruct/Input","Llama 3.3 70B Instruct: price per million tokens by provider (Input)",0.293,"usd","$0.29","fp8 · Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-llama-3-3-70b-instruct","Output","Cloudflare (fp8)","provider-prices-llama-3-3-70b-instruct/Output","Llama 3.3 70B Instruct: price per million tokens by provider (Output)",2.253,"usd","$2.25","fp8 · Llama 3.3 70B Instruct · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-deepseek-v4-pro","Input","Cloudflare","provider-prices-deepseek-v4-pro/Input","DeepSeek V4 Pro 0423: price per million tokens by provider (Input)",1.15,"usd","$1.15","DeepSeek V4 Pro 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-deepseek-v4-pro","Output","Cloudflare","provider-prices-deepseek-v4-pro/Output","DeepSeek V4 Pro 0423: price per million tokens by provider (Output)",2.55,"usd","$2.55","DeepSeek V4 Pro 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-deepseek-v4-pro","Cache read","Cloudflare","provider-prices-deepseek-v4-pro/Cache read","DeepSeek V4 Pro 0423: price per million tokens by provider (Cache read)",0.2,"usd","$0.20","DeepSeek V4 Pro 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-deepseek-v4-flash","Input","Cloudflare","provider-prices-deepseek-v4-flash/Input","DeepSeek V4 Flash 0423: price per million tokens by provider (Input)",0.44,"usd","$0.44","DeepSeek V4 Flash 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-deepseek-v4-flash","Output","Cloudflare","provider-prices-deepseek-v4-flash/Output","DeepSeek V4 Flash 0423: price per million tokens by provider (Output)",1.32,"usd","$1.32","DeepSeek V4 Flash 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-deepseek-v4-flash","Cache read","Cloudflare","provider-prices-deepseek-v4-flash/Cache read","DeepSeek V4 Flash 0423: price per million tokens by provider (Cache read)",0.014,"usd","$0.014","DeepSeek V4 Flash 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-glm-5-3","Input","Cloudflare","provider-prices-glm-5-3/Input","GLM 5.3: price per million tokens by provider (Input)",1.4,"usd","$1.40","GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-glm-5-3","Output","Cloudflare","provider-prices-glm-5-3/Output","GLM 5.3: price per million tokens by provider (Output)",4.4,"usd","$4.40","GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-glm-5-3","Cache read","Cloudflare","provider-prices-glm-5-3/Cache read","GLM 5.3: price per million tokens by provider (Cache read)",0.26,"usd","$0.26","GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"]]}],["siliconflow","SiliconFlow","SiliconFlow","provider","An inference provider listed on OpenRouter. Its prices here are the ones OpenRouter’s public API reported for its endpoints.",["SiliconFlow"],{"$k":["studySlug","chartId","series","point","metric","label","value","unit","display","context"],"$r":[["inference-provider-index","provider-prices-gpt-oss-120b","Input","SiliconFlow (fp8)","provider-prices-gpt-oss-120b/Input","gpt-oss-120b: price per million tokens by provider (Input)",0.15,"usd","$0.15","fp8 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-gpt-oss-120b","Output","SiliconFlow (fp8)","provider-prices-gpt-oss-120b/Output","gpt-oss-120b: price per million tokens by provider (Output)",0.6,"usd","$0.60","fp8 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-gpt-oss-120b","Cache read","SiliconFlow (fp8)","provider-prices-gpt-oss-120b/Cache read","gpt-oss-120b: price per million tokens by provider (Cache read)",0.075,"usd","$0.075","fp8 · gpt-oss-120b · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-deepseek-v4-pro","Input","SiliconFlow (fp8)","provider-prices-deepseek-v4-pro/Input","DeepSeek V4 Pro 0423: price per million tokens by provider (Input)",1.50162,"usd","$1.50","fp8 · DeepSeek V4 Pro 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-deepseek-v4-pro","Output","SiliconFlow (fp8)","provider-prices-deepseek-v4-pro/Output","DeepSeek V4 Pro 0423: price per million tokens by provider (Output)",3.135,"usd","$3.13","fp8 · DeepSeek V4 Pro 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-deepseek-v4-pro","Cache read","SiliconFlow (fp8)","provider-prices-deepseek-v4-pro/Cache read","DeepSeek V4 Pro 0423: price per million tokens by provider (Cache read)",0.135,"usd","$0.14","fp8 · DeepSeek V4 Pro 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-deepseek-v4-flash","Input","SiliconFlow (fp8)","provider-prices-deepseek-v4-flash/Input","DeepSeek V4 Flash 0423: price per million tokens by provider (Input)",0.13,"usd","$0.13","fp8 · DeepSeek V4 Flash 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-deepseek-v4-flash","Output","SiliconFlow (fp8)","provider-prices-deepseek-v4-flash/Output","DeepSeek V4 Flash 0423: price per million tokens by provider (Output)",0.28,"usd","$0.28","fp8 · DeepSeek V4 Flash 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-deepseek-v4-flash","Cache read","SiliconFlow (fp8)","provider-prices-deepseek-v4-flash/Cache read","DeepSeek V4 Flash 0423: price per million tokens by provider (Cache read)",0.028,"usd","$0.028","fp8 · DeepSeek V4 Flash 0423 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-glm-5-3","Input","SiliconFlow (fp8)","provider-prices-glm-5-3/Input","GLM 5.3: price per million tokens by provider (Input)",0.7,"usd","$0.70","fp8 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-glm-5-3","Output","SiliconFlow (fp8)","provider-prices-glm-5-3/Output","GLM 5.3: price per million tokens by provider (Output)",2.2,"usd","$2.20","fp8 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"],["inference-provider-index","provider-prices-glm-5-3","Cache read","SiliconFlow (fp8)","provider-prices-glm-5-3/Cache read","GLM 5.3: price per million tokens by provider (Cache read)",0.13,"usd","$0.13","fp8 · GLM 5.3 · reported by OpenRouter’s public API, snapshot 2026-10-06"]]}]]}