{"$k":["studySlug","chartId","series","point","metric","label","value","unit","display","n","ci","spanKind","context","calculation","range","polarity","statId"],"$r":[["swe-bench-opus-vs-sonnet","swebench-opus-sonnet-resolved","Resolved","Claude Sonnet 5.5 (Agent, older builds)","swebench-opus-sonnet-resolved","Resolved on the same 3 SWE-bench Verified instances (interim)",0.3333,"rate","33% (1/3)",3,[0.0615,0.7923],"ci95","Agent · older builds · SWE-bench Verified, interim paired probe","\u0001","\u0001","\u0001","\u0001"],["swe-bench-opus-vs-sonnet","swebench-opus-sonnet-cost-per-attempt","List-price cost per attempt","Claude Sonnet 5.5 (Agent, older builds)","swebench-opus-sonnet-cost-per-attempt","List-price cost per attempt (calculation)",2.88,"usd","$2.88",3,"\u0001","\u0001","Agent · older builds · SWE-bench Verified, interim paired probe",true,"\u0001","\u0001","\u0001"],["swe-bench-opus-vs-sonnet","swebench-opus-sonnet-minutes","Worker minutes per attempt","Claude Sonnet 5.5 (Agent, older builds)","swebench-opus-sonnet-minutes","Worker time per attempt",9.37,"minutes","9.4 min",3,"\u0001","minmax","Agent · older builds · SWE-bench Verified, interim paired probe","\u0001",[4.74,15],"\u0001","\u0001"],["model-head-to-head","h2h-pass-rate","Pass rate","Claude Sonnet 5.5 · Claude Code","h2h-pass-rate","Pass rate on five validated tasks",0.8,"rate","80% (12/15)",15,[0.5481,0.9295],"ci95","Claude Code · five short validated tasks","\u0001","\u0001","\u0001","\u0001"],["model-head-to-head","h2h-total-latency","Total time per call","Claude Sonnet 5.5 · Claude Code","h2h-total-latency","Total time per call",2.31,"seconds","2.31 s",15,"\u0001","minmax","Claude Code · five short validated tasks","\u0001",[2.17,7.73],"\u0001","\u0001"],["model-head-to-head","h2h-first-useful-latency","Time to first useful output","Claude Sonnet 5.5 · Claude Code","h2h-first-useful-latency","Time to first useful output",1.56,"seconds","1.56 s",15,"\u0001","minmax","Claude Code · five short validated tasks","\u0001",[0.99,6.39],"\u0001","\u0001"],["model-head-to-head","h2h-input-tokens","Cache read","Claude Sonnet 5.5 · Claude Code","h2h-input-tokens/Cache read","Input tokens per call: what the CLI sends (Cache read)",1401,"tokens","1,401",15,"\u0001","\u0001","Claude Code · five short validated tasks","\u0001","\u0001","\u0001","\u0001"],["model-head-to-head","h2h-input-tokens","Other input","Claude Sonnet 5.5 · Claude Code","h2h-input-tokens/Other input","Input tokens per call: what the CLI sends (Other input)",685,"tokens","685",15,"\u0001","\u0001","Claude Code · five short validated tasks","\u0001","\u0001","\u0001","\u0001"],["model-head-to-head","h2h-output-tokens","Output tokens","Claude Sonnet 5.5 · Claude Code","h2h-output-tokens/Output tokens","Output tokens per call (Output tokens)",107,"tokens","107",15,"\u0001","\u0001","Claude Code · five short validated tasks","\u0001","\u0001","\u0001","\u0001"],["model-head-to-head","h2h-list-price-per-call","Cost per call","Claude Sonnet 5.5 · Claude Code","h2h-list-price-per-call","List-price cost per call (calculation)",0.0036,"usd","$0.0036",15,"\u0001","minmax","Claude Code · five short validated tasks",true,[0.00342,0.01021],"\u0001","\u0001"],["model-head-to-head","h2h-cost-per-pass","Cost per pass","Claude Sonnet 5.5 · Claude Code","h2h-cost-per-pass","List-price cost per passing answer (calculation)",0.00624,"usd","$0.0062",15,"\u0001","\u0001","Claude Code · five short validated tasks",true,"\u0001","\u0001","\u0001"],["hard-model-head-to-head","hard-h2h-pass-rate","Strict pass","Claude Sonnet 5.5 · Claude Code","hard-h2h-pass-rate/Strict pass","Pass rate on eight hard tasks (Strict pass)",1,"rate","100% (24/24)",24,[0.862,1],"ci95","Claude Code · eight hard validated tasks","\u0001","\u0001","\u0001","\u0001"],["hard-model-head-to-head","hard-h2h-pass-rate","Lenient (format misses counted)","Claude Sonnet 5.5 · Claude Code","hard-h2h-pass-rate/Lenient (format misses counted)","Pass rate on eight hard tasks (Lenient (format misses counted))",1,"rate","100% (24/24)",24,[0.862,1],"ci95","Claude Code · eight hard validated tasks","\u0001","\u0001","\u0001","\u0001"],["hard-model-head-to-head","hard-h2h-total-latency","Total time per call on hard tasks (separate batches)","Claude Sonnet 5.5 · Claude Code","hard-h2h-total-latency","Total time per call on hard tasks (separate batches)",7.75,"seconds","7.75 s",24,"\u0001","minmax","Claude Code · eight hard validated tasks","\u0001",[2.26,34.79],"\u0001","\u0001"],["hard-model-head-to-head","hard-h2h-first-useful-latency","Time to first useful output on hard tasks","Claude Sonnet 5.5 · Claude Code","hard-h2h-first-useful-latency","Time to first useful output on hard tasks",5.95,"seconds","5.95 s",24,"\u0001","minmax","Claude Code · eight hard validated tasks","\u0001",[0.86,30.57],"\u0001","\u0001"],["hard-model-head-to-head","hard-h2h-output-tokens","Output tokens","Claude Sonnet 5.5 · Claude Code","hard-h2h-output-tokens/Output tokens","Output tokens per call on hard tasks (Output tokens)",1050,"tokens","1,050",24,"\u0001","\u0001","Claude Code · eight hard validated tasks","\u0001","\u0001","\u0001","\u0001"],["hard-model-head-to-head","hard-h2h-cost-per-pass","Cost per strict pass","Claude Sonnet 5.5 · Claude Code","hard-h2h-cost-per-pass","List-price cost per strict pass on hard tasks (calculation)",0.01435,"usd","$0.014",24,"\u0001","\u0001","Claude Code · eight hard validated tasks",true,"\u0001","\u0001","\u0001"],["coding-agents-head-to-head","coding-agents-pass-rate","Passed every hidden check","Claude Sonnet 5.5 · Claude Code","coding-agents-pass-rate","Coding sessions that passed every hidden check",1,"rate","100% (12/12)",12,[0.7575,1],"ci95","Claude Code · six small repository tasks with hidden tests","\u0001","\u0001","\u0001","\u0001"],["coding-agents-head-to-head","coding-agents-wall-time","Wall time per session","Claude Sonnet 5.5 · Claude Code","coding-agents-wall-time","Time per coding session",23.1,"seconds","23.1 s",12,"\u0001","minmax","Claude Code · six small repository tasks with hidden tests","\u0001",[18.7,44.5],"\u0001","\u0001"],["coding-agents-head-to-head","coding-agents-tool-calls","Tool calls per session","Claude Sonnet 5.5 · Claude Code","coding-agents-tool-calls","Tool calls per coding session",7.5,"calls","7.5",12,"\u0001","minmax","Claude Code · six small repository tasks with hidden tests","\u0001",[3,14],"\u0001","\u0001"],["coding-agents-head-to-head","coding-agents-cost-per-pass","List-price cost per pass","Claude Sonnet 5.5 · Claude Code","coding-agents-cost-per-pass","List-price cost per passing coding session (calculation)",0.085,"usd","$0.085",12,"\u0001","\u0001","Claude Code · six small repository tasks with hidden tests",true,"\u0001","\u0001","\u0001"],["effort-ladder","effort-ladder-pass-rate","Strict pass","Claude Sonnet 5.5 (low) · Claude Code","effort-ladder-pass-rate","Strict pass rate by effort on eight hard tasks",1,"rate","100% (16/16)",16,[0.8064,1],"ci95","Claude Code · effort low · eight hard validated tasks, effort ladder","\u0001","\u0001","\u0001","\u0001"],["effort-ladder","effort-ladder-pass-rate","Strict pass","Claude Sonnet 5.5 (medium) · Claude Code","effort-ladder-pass-rate","Strict pass rate by effort on eight hard tasks",1,"rate","100% (16/16)",16,[0.8064,1],"ci95","Claude Code · effort medium · eight hard validated tasks, effort ladder","\u0001","\u0001","\u0001","\u0001"],["effort-ladder","effort-ladder-pass-rate","Strict pass","Claude Sonnet 5.5 (high) · Claude Code","effort-ladder-pass-rate","Strict pass rate by effort on eight hard tasks",1,"rate","100% (16/16)",16,[0.8064,1],"ci95","Claude Code · effort high · eight hard validated tasks, effort ladder","\u0001","\u0001","\u0001","\u0001"],["effort-ladder","effort-ladder-pass-rate","Strict pass","Claude Sonnet 5.5 · Claude Code","effort-ladder-pass-rate","Strict pass rate by effort on eight hard tasks",1,"rate","100% (16/16)",16,[0.8064,1],"ci95","Claude Code · eight hard validated tasks, effort ladder","\u0001","\u0001","\u0001","\u0001"],["effort-ladder","effort-ladder-total-latency","Total time per call by effort on hard tasks","Claude Sonnet 5.5 (low) · Claude Code","effort-ladder-total-latency","Total time per call by effort on hard tasks",5.82,"seconds","5.82 s",16,"\u0001","minmax","Claude Code · effort low · eight hard validated tasks, effort ladder","\u0001",[2.78,19.96],"\u0001","\u0001"],["effort-ladder","effort-ladder-total-latency","Total time per call by effort on hard tasks","Claude Sonnet 5.5 (medium) · Claude Code","effort-ladder-total-latency","Total time per call by effort on hard tasks",7.63,"seconds","7.63 s",16,"\u0001","minmax","Claude Code · effort medium · eight hard validated tasks, effort ladder","\u0001",[2.71,24.01],"\u0001","\u0001"],["effort-ladder","effort-ladder-total-latency","Total time per call by effort on hard tasks","Claude Sonnet 5.5 (high) · Claude Code","effort-ladder-total-latency","Total time per call by effort on hard tasks",8.81,"seconds","8.81 s",16,"\u0001","minmax","Claude Code · effort high · eight hard validated tasks, effort ladder","\u0001",[2.93,35.81],"\u0001","\u0001"],["effort-ladder","effort-ladder-total-latency","Total time per call by effort on hard tasks","Claude Sonnet 5.5 · Claude Code","effort-ladder-total-latency","Total time per call by effort on hard tasks",7.97,"seconds","7.97 s",16,"\u0001","minmax","Claude Code · eight hard validated tasks, effort ladder","\u0001",[2.26,21.61],"\u0001","\u0001"],["effort-ladder","effort-ladder-output-tokens","Output tokens","Claude Sonnet 5.5 (low) · Claude Code","effort-ladder-output-tokens/Output tokens","Output tokens per call by effort on hard tasks (Output tokens)",667,"tokens","667",16,"\u0001","\u0001","Claude Code · effort low · eight hard validated tasks, effort ladder","\u0001","\u0001","\u0001","\u0001"],["effort-ladder","effort-ladder-output-tokens","Output tokens","Claude Sonnet 5.5 (medium) · Claude Code","effort-ladder-output-tokens/Output tokens","Output tokens per call by effort on hard tasks (Output tokens)",770,"tokens","770",16,"\u0001","\u0001","Claude Code · effort medium · eight hard validated tasks, effort ladder","\u0001","\u0001","\u0001","\u0001"],["effort-ladder","effort-ladder-output-tokens","Output tokens","Claude Sonnet 5.5 (high) · Claude Code","effort-ladder-output-tokens/Output tokens","Output tokens per call by effort on hard tasks (Output tokens)",1192,"tokens","1,192",16,"\u0001","\u0001","Claude Code · effort high · eight hard validated tasks, effort ladder","\u0001","\u0001","\u0001","\u0001"],["effort-ladder","effort-ladder-output-tokens","Output tokens","Claude Sonnet 5.5 · Claude Code","effort-ladder-output-tokens/Output tokens","Output tokens per call by effort on hard tasks (Output tokens)",1054,"tokens","1,054",16,"\u0001","\u0001","Claude Code · eight hard validated tasks, effort ladder","\u0001","\u0001","\u0001","\u0001"],["effort-ladder","effort-ladder-cost-per-pass","Cost per strict pass","Claude Sonnet 5.5 (low) · Claude Code","effort-ladder-cost-per-pass","List-price cost per strict pass by effort (calculation)",0.01219,"usd","$0.012",16,"\u0001","\u0001","Claude Code · effort low · eight hard validated tasks, effort ladder",true,"\u0001","\u0001","\u0001"],["effort-ladder","effort-ladder-cost-per-pass","Cost per strict pass","Claude Sonnet 5.5 (medium) · Claude Code","effort-ladder-cost-per-pass","List-price cost per strict pass by effort (calculation)",0.01352,"usd","$0.014",16,"\u0001","\u0001","Claude Code · effort medium · eight hard validated tasks, effort ladder",true,"\u0001","\u0001","\u0001"],["effort-ladder","effort-ladder-cost-per-pass","Cost per strict pass","Claude Sonnet 5.5 (high) · Claude Code","effort-ladder-cost-per-pass","List-price cost per strict pass by effort (calculation)",0.01671,"usd","$0.017",16,"\u0001","\u0001","Claude Code · effort high · eight hard validated tasks, effort ladder",true,"\u0001","\u0001","\u0001"],["effort-ladder","effort-ladder-cost-per-pass","Cost per strict pass","Claude Sonnet 5.5 · Claude Code","effort-ladder-cost-per-pass","List-price cost per strict pass by effort (calculation)",0.01398,"usd","$0.014",16,"\u0001","\u0001","Claude Code · eight hard validated tasks, effort ladder",true,"\u0001","\u0001","\u0001"],["caching-consistency","caching-cost-with-without","With the cache, as recorded","Claude Sonnet 5.5 · Claude Code","caching-cost-with-without/With the cache, as recorded","List-price cost of 5-question sessions with and without the cache (calculation) (With the cache, as recorded)",0.135003,"usd","$0.14",15,"\u0001","\u0001","Claude Code · calculation: 5-turn cached sessions over a fixed ledger",true,"\u0001","\u0001","\u0001"],["caching-consistency","caching-cost-with-without","Without a cache: every input token at the input price","Claude Sonnet 5.5 · Claude Code","caching-cost-with-without/Without a cache: every input token at the input price","List-price cost of 5-question sessions with and without the cache (calculation) (Without a cache: every input token at the input price)",0.269788,"usd","$0.27",15,"\u0001","\u0001","Claude Code · calculation: 5-turn cached sessions over a fixed ledger",true,"\u0001","\u0001","\u0001"],["caching-consistency","caching-latency-first-vs-later","Turn 1 (writes the ledger to the cache)","Claude Sonnet 5.5 · Claude Code","caching-latency-first-vs-later/Turn 1 (writes the ledger to the cache)","Time per turn: first turn vs later turns in a cached session (Turn 1 (writes the ledger to the cache))",1.64,"seconds","1.64 s",3,"\u0001","minmax","Claude Code · 5-turn cached sessions over a fixed ledger","\u0001",[1.58,1.79],"\u0001","\u0001"],["caching-consistency","caching-latency-first-vs-later","Turns 2-5 (read the ledger from the cache)","Claude Sonnet 5.5 · Claude Code","caching-latency-first-vs-later/Turns 2-5 (read the ledger from the cache)","Time per turn: first turn vs later turns in a cached session (Turns 2-5 (read the ledger from the cache))",1.61,"seconds","1.61 s",12,"\u0001","minmax","Claude Code · 5-turn cached sessions over a fixed ledger","\u0001",[1.35,5.63],"\u0001","\u0001"],["caching-consistency","consistency-pass-rate","Exact number","Claude Sonnet 5.5 · Claude Code","consistency-pass-rate/Exact number","Same prompt, 10 times: strict pass rate (Exact number)",1,"rate","100% (10/10)",10,[0.7225,1],"ci95","Claude Code · same prompt repeated 10 times","\u0001","\u0001","\u0001","\u0001"],["caching-consistency","consistency-pass-rate","JSON object","Claude Sonnet 5.5 · Claude Code","consistency-pass-rate/JSON object","Same prompt, 10 times: strict pass rate (JSON object)",1,"rate","100% (10/10)",10,[0.7225,1],"ci95","Claude Code · same prompt repeated 10 times","\u0001","\u0001","\u0001","\u0001"],["caching-consistency","consistency-pass-rate","Code fix","Claude Sonnet 5.5 · Claude Code","consistency-pass-rate/Code fix","Same prompt, 10 times: strict pass rate (Code fix)",1,"rate","100% (10/10)",10,[0.7225,1],"ci95","Claude Code · same prompt repeated 10 times","\u0001","\u0001","\u0001","\u0001"],["caching-consistency","consistency-distinct-answers","Exact number","Claude Sonnet 5.5 · Claude Code","consistency-distinct-answers/Exact number","Same prompt, 10 times: how many different answers (Exact number)",1,"count","1",10,"\u0001","\u0001","Claude Code · same prompt repeated 10 times","\u0001","\u0001","\u0001","\u0001"],["caching-consistency","consistency-distinct-answers","JSON object","Claude Sonnet 5.5 · Claude Code","consistency-distinct-answers/JSON object","Same prompt, 10 times: how many different answers (JSON object)",1,"count","1",10,"\u0001","\u0001","Claude Code · same prompt repeated 10 times","\u0001","\u0001","\u0001","\u0001"],["caching-consistency","consistency-distinct-answers","Code fix","Claude Sonnet 5.5 · Claude Code","consistency-distinct-answers/Code fix","Same prompt, 10 times: how many different answers (Code fix)",3,"count","3",10,"\u0001","\u0001","Claude Code · same prompt repeated 10 times","\u0001","\u0001","\u0001","\u0001"],["caching-consistency","consistency-latency-spread","Exact number","Claude Sonnet 5.5 · Claude Code","consistency-latency-spread/Exact number","Same prompt, 10 times: time per call (Exact number)",6.89,"seconds","6.89 s",10,"\u0001","minmax","Claude Code · same prompt repeated 10 times","\u0001",[5.81,7.81],"\u0001","\u0001"],["caching-consistency","consistency-latency-spread","JSON object","Claude Sonnet 5.5 · Claude Code","consistency-latency-spread/JSON object","Same prompt, 10 times: time per call (JSON object)",2.89,"seconds","2.89 s",10,"\u0001","minmax","Claude Code · same prompt repeated 10 times","\u0001",[2.68,5.3],"\u0001","\u0001"],["caching-consistency","consistency-latency-spread","Code fix","Claude Sonnet 5.5 · Claude Code","consistency-latency-spread/Code fix","Same prompt, 10 times: time per call (Code fix)",2.67,"seconds","2.67 s",10,"\u0001","minmax","Claude Code · same prompt repeated 10 times","\u0001",[2.32,4.34],"\u0001","\u0001"],["agent-memory","memory-full-pass","Claude Sonnet 5.5","No memory","memory-full-pass/No memory","Full pass rate by kind of memory: No memory",0.6,"rate","60% (9/15)",15,[0.3575,0.8018],"ci95","","\u0001","\u0001","\u0001","\u0001"],["agent-memory","memory-full-pass","Claude Sonnet 5.5","/init CLAUDE.md","memory-full-pass//init CLAUDE.md","Full pass rate by kind of memory: /init CLAUDE.md",0.6,"rate","60% (9/15)",15,[0.3575,0.8018],"ci95","","\u0001","\u0001","\u0001","\u0001"],["agent-memory","memory-full-pass","Claude Sonnet 5.5","Curated, 11 lines","memory-full-pass/Curated, 11 lines","Full pass rate by kind of memory: Curated, 11 lines",1,"rate","100% (15/15)",15,[0.7961,1],"ci95","","\u0001","\u0001","\u0001","\u0001"],["agent-memory","memory-full-pass","Claude Sonnet 5.5","Raw notes, 60 lines","memory-full-pass/Raw notes, 60 lines","Full pass rate by kind of memory: Raw notes, 60 lines",0.9333,"rate","93% (14/15)",15,[0.7018,0.9881],"ci95","","\u0001","\u0001","\u0001","\u0001"],["agent-memory","memory-full-pass","Claude Sonnet 5.5","Dreamed notes","memory-full-pass/Dreamed notes","Full pass rate by kind of memory: Dreamed notes",1,"rate","100% (15/15)",15,[0.7961,1],"ci95","","\u0001","\u0001","\u0001","\u0001"],["agent-memory","memory-full-pass","Claude Sonnet 5.5","Handbook, 210 lines","memory-full-pass/Handbook, 210 lines","Full pass rate by kind of memory: Handbook, 210 lines",1,"rate","100% (15/15)",15,[0.7961,1],"ci95","","\u0001","\u0001","\u0001","\u0001"],["agent-memory","memory-full-pass","Claude Sonnet 5.5","Stop hook only","memory-full-pass/Stop hook only","Full pass rate by kind of memory: Stop hook only",0.8,"rate","80% (12/15)",15,[0.5481,0.9295],"ci95","","\u0001","\u0001","\u0001","\u0001"],["agent-memory","memory-full-pass","Claude Sonnet 5.5","Curated + hook","memory-full-pass/Curated + hook","Full pass rate by kind of memory: Curated + hook",1,"rate","100% (15/15)",15,[0.7961,1],"ci95","","\u0001","\u0001","\u0001","\u0001"],["agent-memory","memory-team-knowledge-by-model","Claude Sonnet 5.5","No memory","memory-team-knowledge-by-model/No memory","Team knowledge followed, Sonnet vs Haiku: No memory",0.4,"rate","40% (6/15)",15,[0.1982,0.6425],"ci95","","\u0001","\u0001","\u0001","\u0001"],["agent-memory","memory-team-knowledge-by-model","Claude Sonnet 5.5","/init CLAUDE.md","memory-team-knowledge-by-model//init CLAUDE.md","Team knowledge followed, Sonnet vs Haiku: /init CLAUDE.md",0.6667,"rate","67% (10/15)",15,[0.4171,0.8482],"ci95","","\u0001","\u0001","\u0001","\u0001"],["agent-memory","memory-team-knowledge-by-model","Claude Sonnet 5.5","Curated, 11 lines","memory-team-knowledge-by-model/Curated, 11 lines","Team knowledge followed, Sonnet vs Haiku: Curated, 11 lines",1,"rate","100% (15/15)",15,[0.7961,1],"ci95","","\u0001","\u0001","\u0001","\u0001"],["agent-memory","memory-team-knowledge-by-model","Claude Sonnet 5.5","Raw notes, 60 lines","memory-team-knowledge-by-model/Raw notes, 60 lines","Team knowledge followed, Sonnet vs Haiku: Raw notes, 60 lines",1,"rate","100% (15/15)",15,[0.7961,1],"ci95","","\u0001","\u0001","\u0001","\u0001"],["agent-memory","memory-team-knowledge-by-model","Claude Sonnet 5.5","Dreamed notes","memory-team-knowledge-by-model/Dreamed notes","Team knowledge followed, Sonnet vs Haiku: Dreamed notes",1,"rate","100% (15/15)",15,[0.7961,1],"ci95","","\u0001","\u0001","\u0001","\u0001"],["agent-memory","memory-team-knowledge-by-model","Claude Sonnet 5.5","Handbook, 210 lines","memory-team-knowledge-by-model/Handbook, 210 lines","Team knowledge followed, Sonnet vs Haiku: Handbook, 210 lines",1,"rate","100% (15/15)",15,[0.7961,1],"ci95","","\u0001","\u0001","\u0001","\u0001"],["agent-memory","memory-team-knowledge-by-model","Claude Sonnet 5.5","Stop hook only","memory-team-knowledge-by-model/Stop hook only","Team knowledge followed, Sonnet vs Haiku: Stop hook only",0.6667,"rate","67% (10/15)",15,[0.4171,0.8482],"ci95","","\u0001","\u0001","\u0001","\u0001"],["agent-memory","memory-team-knowledge-by-model","Claude Sonnet 5.5","Curated + hook","memory-team-knowledge-by-model/Curated + hook","Team knowledge followed, Sonnet vs Haiku: Curated + hook",1,"rate","100% (15/15)",15,[0.7961,1],"ci95","","\u0001","\u0001","\u0001","\u0001"],["agent-memory","memory-broken-test-command","Claude Sonnet 5.5","No memory","memory-broken-test-command/No memory","A stale README command: who still ran it?: No memory",0.8,"rate","80% (12/15)",15,[0.5481,0.9295],"ci95","","\u0001","\u0001","lower","\u0001"],["agent-memory","memory-broken-test-command","Claude Sonnet 5.5","/init CLAUDE.md","memory-broken-test-command//init CLAUDE.md","A stale README command: who still ran it?: /init CLAUDE.md",0.8667,"rate","87% (13/15)",15,[0.6212,0.9626],"ci95","","\u0001","\u0001","lower","\u0001"],["agent-memory","memory-broken-test-command","Claude Sonnet 5.5","Curated, 11 lines","memory-broken-test-command/Curated, 11 lines","A stale README command: who still ran it?: Curated, 11 lines",0,"rate","0% (0/15)",15,[0,0.2039],"ci95","","\u0001","\u0001","lower","\u0001"],["agent-memory","memory-broken-test-command","Claude Sonnet 5.5","Raw notes, 60 lines","memory-broken-test-command/Raw notes, 60 lines","A stale README command: who still ran it?: Raw notes, 60 lines",0,"rate","0% (0/15)",15,[0,0.2039],"ci95","","\u0001","\u0001","lower","\u0001"],["agent-memory","memory-broken-test-command","Claude Sonnet 5.5","Dreamed notes","memory-broken-test-command/Dreamed notes","A stale README command: who still ran it?: Dreamed notes",0,"rate","0% (0/15)",15,[0,0.2039],"ci95","","\u0001","\u0001","lower","\u0001"],["agent-memory","memory-broken-test-command","Claude Sonnet 5.5","Handbook, 210 lines","memory-broken-test-command/Handbook, 210 lines","A stale README command: who still ran it?: Handbook, 210 lines",0,"rate","0% (0/15)",15,[0,0.2039],"ci95","","\u0001","\u0001","lower","\u0001"],["agent-memory","memory-broken-test-command","Claude Sonnet 5.5","Stop hook only","memory-broken-test-command/Stop hook only","A stale README command: who still ran it?: Stop hook only",0.6,"rate","60% (9/15)",15,[0.3575,0.8018],"ci95","","\u0001","\u0001","lower","\u0001"],["agent-memory","memory-broken-test-command","Claude Sonnet 5.5","Curated + hook","memory-broken-test-command/Curated + hook","A stale README command: who still ran it?: Curated + hook",0,"rate","0% (0/15)",15,[0,0.2039],"ci95","","\u0001","\u0001","lower","\u0001"],["agent-memory","memory-cost-per-full-pass","Claude Sonnet 5.5","No memory","memory-cost-per-full-pass/No memory","List-price cost per fully correct result (calculation): No memory",0.1386,"usd","$0.14",9,"\u0001","\u0001","",true,"\u0001","\u0001","\u0001"],["agent-memory","memory-cost-per-full-pass","Claude Sonnet 5.5","/init CLAUDE.md","memory-cost-per-full-pass//init CLAUDE.md","List-price cost per fully correct result (calculation): /init CLAUDE.md",0.1359,"usd","$0.14",9,"\u0001","\u0001","",true,"\u0001","\u0001","\u0001"],["agent-memory","memory-cost-per-full-pass","Claude Sonnet 5.5","Curated, 11 lines","memory-cost-per-full-pass/Curated, 11 lines","List-price cost per fully correct result (calculation): Curated, 11 lines",0.0818,"usd","$0.082",15,"\u0001","\u0001","",true,"\u0001","\u0001","\u0001"],["agent-memory","memory-cost-per-full-pass","Claude Sonnet 5.5","Raw notes, 60 lines","memory-cost-per-full-pass/Raw notes, 60 lines","List-price cost per fully correct result (calculation): Raw notes, 60 lines",0.1009,"usd","$0.10",14,"\u0001","\u0001","",true,"\u0001","\u0001","\u0001"],["agent-memory","memory-cost-per-full-pass","Claude Sonnet 5.5","Dreamed notes","memory-cost-per-full-pass/Dreamed notes","List-price cost per fully correct result (calculation): Dreamed notes",0.0896,"usd","$0.090",15,"\u0001","\u0001","",true,"\u0001","\u0001","\u0001"],["agent-memory","memory-cost-per-full-pass","Claude Sonnet 5.5","Handbook, 210 lines","memory-cost-per-full-pass/Handbook, 210 lines","List-price cost per fully correct result (calculation): Handbook, 210 lines",0.1006,"usd","$0.10",15,"\u0001","\u0001","",true,"\u0001","\u0001","\u0001"],["agent-memory","memory-cost-per-full-pass","Claude Sonnet 5.5","Stop hook only","memory-cost-per-full-pass/Stop hook only","List-price cost per fully correct result (calculation): Stop hook only",0.1278,"usd","$0.13",12,"\u0001","\u0001","",true,"\u0001","\u0001","\u0001"],["agent-memory","memory-cost-per-full-pass","Claude Sonnet 5.5","Curated + hook","memory-cost-per-full-pass/Curated + hook","List-price cost per fully correct result (calculation): Curated + hook",0.0843,"usd","$0.084",15,"\u0001","\u0001","",true,"\u0001","\u0001","\u0001"],["agent-memory","memory-wall-time","Claude Sonnet 5.5","No memory","memory-wall-time/No memory","Time per session: No memory",18,"seconds","18.0 s",15,"\u0001","\u0001","","\u0001","\u0001","\u0001","\u0001"],["agent-memory","memory-wall-time","Claude Sonnet 5.5","/init CLAUDE.md","memory-wall-time//init CLAUDE.md","Time per session: /init CLAUDE.md",19,"seconds","19.0 s",15,"\u0001","\u0001","","\u0001","\u0001","\u0001","\u0001"],["agent-memory","memory-wall-time","Claude Sonnet 5.5","Curated, 11 lines","memory-wall-time/Curated, 11 lines","Time per session: Curated, 11 lines",21.9,"seconds","21.9 s",15,"\u0001","\u0001","","\u0001","\u0001","\u0001","\u0001"],["agent-memory","memory-wall-time","Claude Sonnet 5.5","Raw notes, 60 lines","memory-wall-time/Raw notes, 60 lines","Time per session: Raw notes, 60 lines",26.8,"seconds","26.8 s",15,"\u0001","\u0001","","\u0001","\u0001","\u0001","\u0001"],["agent-memory","memory-wall-time","Claude Sonnet 5.5","Dreamed notes","memory-wall-time/Dreamed notes","Time per session: Dreamed notes",27.2,"seconds","27.2 s",15,"\u0001","\u0001","","\u0001","\u0001","\u0001","\u0001"],["agent-memory","memory-wall-time","Claude Sonnet 5.5","Handbook, 210 lines","memory-wall-time/Handbook, 210 lines","Time per session: Handbook, 210 lines",23.6,"seconds","23.6 s",15,"\u0001","\u0001","","\u0001","\u0001","\u0001","\u0001"],["agent-memory","memory-wall-time","Claude Sonnet 5.5","Stop hook only","memory-wall-time/Stop hook only","Time per session: Stop hook only",27.3,"seconds","27.3 s",15,"\u0001","\u0001","","\u0001","\u0001","\u0001","\u0001"],["agent-memory","memory-wall-time","Claude Sonnet 5.5","Curated + hook","memory-wall-time/Curated + hook","Time per session: Curated + hook",22,"seconds","22.0 s",15,"\u0001","\u0001","","\u0001","\u0001","\u0001","\u0001"],["routing-jev-vs-llm","routing-exact-decisions","Exact rate","Claude Sonnet 5.5","routing-exact-decisions","Typed routing decisions answered exactly right",0.939,"rate","94% (77/82)",82,[0.8651,0.9737],"ci95","typed routing decisions · via Claude Code","\u0001","\u0001","\u0001","\u0001"],["routing-jev-vs-llm","routing-key-accuracy","Key accuracy","Claude Sonnet 5.5","routing-key-accuracy","Per-question accuracy",0.9742,"rate","97% (189/194)",194,[0.9411,0.9889],"ci95","typed routing decisions · via Claude Code","\u0001","\u0001","\u0001","\u0001"],["routing-jev-vs-llm","routing-exact-by-decision","Claude Sonnet 5.5","Failure class","routing-exact-by-decision/Failure class","Exact rate by decision type: Failure class",1,"rate","100% (18/18)",18,[0.8241,1],"ci95","typed routing decisions · via Claude Code","\u0001","\u0001","\u0001","\u0001"],["routing-jev-vs-llm","routing-exact-by-decision","Claude Sonnet 5.5","Message intent","routing-exact-by-decision/Message intent","Exact rate by decision type: Message intent",1,"rate","100% (20/20)",20,[0.8389,1],"ci95","typed routing decisions · via Claude Code","\u0001","\u0001","\u0001","\u0001"],["routing-jev-vs-llm","routing-exact-by-decision","Claude Sonnet 5.5","Is it a rule?","routing-exact-by-decision/Is it a rule?","Exact rate by decision type: Is it a rule?",1,"rate","100% (12/12)",12,[0.7575,1],"ci95","typed routing decisions · via Claude Code","\u0001","\u0001","\u0001","\u0001"],["routing-jev-vs-llm","routing-exact-by-decision","Claude Sonnet 5.5","Context shape","routing-exact-by-decision/Context shape","Exact rate by decision type: Context shape",0.8438,"rate","84% (27/32)",32,[0.6825,0.9314],"ci95","typed routing decisions · via Claude Code","\u0001","\u0001","\u0001","\u0001"],["routing-jev-vs-llm","routing-cost-per-1000","Cost","Claude Sonnet 5.5","routing-cost-per-1000","Cost per 1,000 routing decisions",4.996,"usd","$5.00",82,"\u0001","\u0001","typed routing decisions · via Claude Code",true,"\u0001","\u0001","\u0001"],["routing-jev-vs-llm","routing-decision-latency","Wall time (CLI)","Claude Sonnet 5.5","routing-decision-latency/Wall time (CLI)","Time per routing decision (Wall time (CLI))",2598,"ms","2,598 ms",82,"\u0001","p50-p95","typed routing decisions · via Claude Code","\u0001",[2598,4298],"\u0001","\u0001"],["routing-jev-vs-llm","routing-decision-latency","Model time (API)","Claude Sonnet 5.5","routing-decision-latency/Model time (API)","Time per routing decision (Model time (API))",1599,"ms","1,599 ms",82,"\u0001","p50-p95","typed routing decisions · via Claude Code","\u0001",[1599,2574],"\u0001","\u0001"],["routing-overhead","router-overhead-decision-latency","Decision time","Claude Sonnet 5.5 (effort low, via Claude Code)","router-overhead-decision-latency","Time to make one routing decision",2597,"ms","2,597 ms",82,"\u0001","p50-p95","effort low · via Claude Code · routing overhead per decision","\u0001",[2597,4298],"\u0001","\u0001"],["routing-overhead","router-overhead-cli-vs-model-time","Model API time","Claude Sonnet 5.5 (effort low, via Claude Code)","router-overhead-cli-vs-model-time/Model API time","Where an LLM router’s time goes: model vs CLI (Model API time)",1596,"ms","1,596 ms",82,"\u0001","p50-p95","effort low · via Claude Code · routing overhead per decision","\u0001",[1596,2583],"\u0001","\u0001"],["routing-overhead","router-overhead-cli-vs-model-time","CLI and harness time","Claude Sonnet 5.5 (effort low, via Claude Code)","router-overhead-cli-vs-model-time/CLI and harness time","Where an LLM router’s time goes: model vs CLI (CLI and harness time)",973,"ms","973 ms",82,"\u0001","p50-p95","effort low · via Claude Code · routing overhead per decision","\u0001",[973,1277],"\u0001","\u0001"],["routing-overhead","router-overhead-completed","Completed","Claude Sonnet 5.5 (effort low, via Claude Code)","router-overhead-completed","Routing calls that returned a decision",1,"rate","100% (82/82)",82,[0.9552,1],"ci95","effort low · via Claude Code · routing overhead per decision","\u0001","\u0001","\u0001","\u0001"],["routing-overhead","router-overhead-cost-list-price","Cost per 1,000 decisions (list price)","Claude Sonnet 5.5 (effort low, via Claude Code)","router-overhead-cost-list-price","Cost per 1,000 routing decisions for the model routers (calculation)",4.996,"usd","$5.00",82,"\u0001","\u0001","effort low · via Claude Code · calculation: Agent’s recorded tokens at this model’s list price · routing overhead per decision",true,"\u0001","\u0001","\u0001"],["routing-overhead","router-overhead-cost-per-1000-tasks","Every model call routed (49.5 per task)","Claude Sonnet 5.5 (effort low, via Claude Code)","router-overhead-cost-per-1000-tasks/Every model call routed (49.5 per task)","Added routing cost per 1,000 tasks (calculation) (Every model call routed (49.5 per task))",247.3,"usd","$247.30","\u0001","\u0001","\u0001","effort low · via Claude Code · calculation per 1,000 tasks from recorded decision counts",true,"\u0001","\u0001","\u0001"],["routing-overhead","router-overhead-cost-per-1000-tasks","Only System One decisions (7 per task)","Claude Sonnet 5.5 (effort low, via Claude Code)","router-overhead-cost-per-1000-tasks/Only System One decisions (7 per task)","Added routing cost per 1,000 tasks (calculation) (Only System One decisions (7 per task))",34.97,"usd","$34.97","\u0001","\u0001","\u0001","effort low · via Claude Code · calculation per 1,000 tasks from recorded decision counts",true,"\u0001","\u0001","\u0001"],["routing-overhead","router-overhead-delay-per-task","Every model call routed (49.5 per task)","Claude Sonnet 5.5 (effort low, via Claude Code)","router-overhead-delay-per-task/Every model call routed (49.5 per task)","Added routing delay per task (calculation) (Every model call routed (49.5 per task))",128.5515,"seconds","128.6 s","\u0001","\u0001","\u0001","effort low · via Claude Code · calculation per task from recorded decision counts, decisions in line",true,"\u0001","\u0001","\u0001"],["routing-overhead","router-overhead-delay-per-task","Only System One decisions (7 per task)","Claude Sonnet 5.5 (effort low, via Claude Code)","router-overhead-delay-per-task/Only System One decisions (7 per task)","Added routing delay per task (calculation) (Only System One decisions (7 per task))",18.179,"seconds","18.2 s","\u0001","\u0001","\u0001","effort low · via Claude Code · calculation per task from recorded decision counts, decisions in line",true,"\u0001","\u0001","\u0001"],["cost-thought-experiments","repriced-cost-per-resolved","Repriced cost per resolved instance","Claude Sonnet 5.5","repriced-cost-per-resolved","Thought experiment: the same tokens at other list prices",3.489,"usd","$3.49","\u0001","\u0001","\u0001","calculation: Agent’s recorded tokens at this model’s list price",true,"\u0001","\u0001","\u0001"],["cost-thought-experiments","prompt-cache-savings","With caching (as recorded)","Claude Sonnet 5.5","prompt-cache-savings/With caching (as recorded)","Thought experiment: what prompt caching saved (With caching (as recorded))",87.23,"usd","$87.23","\u0001","\u0001","\u0001","calculation: Agent’s recorded tokens at this model’s list price",true,"\u0001","\u0001","\u0001"],["cost-thought-experiments","prompt-cache-savings","Without caching","Claude Sonnet 5.5","prompt-cache-savings/Without caching","Thought experiment: what prompt caching saved (Without caching)",343.33,"usd","$343.33","\u0001","\u0001","\u0001","calculation: Agent’s recorded tokens at this model’s list price",true,"\u0001","\u0001","\u0001"],["cli-model-latency-tokens","scheduler-repair-claude-vs-codex","Total time","Claude Code CLI · Sonnet 5.5 · medium","scheduler-repair-claude-vs-codex/Total time","Repairing a scheduler: Claude Code vs Codex vs API (Total time)",15,"seconds","15.0 s",3,"\u0001","minmax","Claude Code CLI · effort medium · scheduler repair, 296 checks, 3 runs","\u0001",[13.89,15.89],"\u0001","\u0001"],["cli-model-latency-tokens","scheduler-repair-claude-vs-codex","First useful output","Claude Code CLI · Sonnet 5.5 · medium","scheduler-repair-claude-vs-codex/First useful output","Repairing a scheduler: Claude Code vs Codex vs API (First useful output)",7.55,"seconds","7.55 s",3,"\u0001","minmax","Claude Code CLI · effort medium · scheduler repair, 296 checks, 3 runs","\u0001",[6.77,7.63],"\u0001","\u0001"],["cli-model-latency-tokens","scheduler-repair-output-tokens","Output tokens","Claude Code CLI · Sonnet 5.5 · medium","scheduler-repair-output-tokens/Output tokens","Output tokens to repair the scheduler (Output tokens)",2227,"tokens","2,227",3,"\u0001","\u0001","Claude Code CLI · effort medium · scheduler repair, 296 checks, 3 runs","\u0001","\u0001","\u0001","\u0001"],["single-call-vs-agent-loop","agent-loop-pass-rate","Strict pass","Claude Sonnet 5.5 (single call) · Claude Code","agent-loop-pass-rate","Strict pass rate: single call vs agent loop on eight hard tasks",1,"rate","100% (24/24)",24,[0.862,1],"ci95","Claude Code · single call","\u0001","\u0001","higher","\u0001"],["single-call-vs-agent-loop","agent-loop-pass-rate","Strict pass","Claude Sonnet 5.5 (agent loop) · Claude Code","agent-loop-pass-rate","Strict pass rate: single call vs agent loop on eight hard tasks",1,"rate","100% (16/16)",16,[0.8064,1],"ci95","Claude Code · agent loop","\u0001","\u0001","higher","\u0001"],["single-call-vs-agent-loop","agent-loop-by-task","Claude Sonnet 5.5 (single call) · Claude Code","Interval merge fix","agent-loop-by-task/Interval merge fix","Strict passes per task: single call vs agent loop: Interval merge fix",1,"rate","100% (3/3)",3,[0.4385,1],"ci95","Claude Code · single call","\u0001","\u0001","higher","\u0001"],["single-call-vs-agent-loop","agent-loop-by-task","Claude Sonnet 5.5 (single call) · Claude Code","DST day-length fix","agent-loop-by-task/DST day-length fix","Strict passes per task: single call vs agent loop: DST day-length fix",1,"rate","100% (3/3)",3,[0.4385,1],"ci95","Claude Code · single call","\u0001","\u0001","higher","\u0001"],["single-call-vs-agent-loop","agent-loop-by-task","Claude Sonnet 5.5 (single call) · Claude Code","CSV parser","agent-loop-by-task/CSV parser","Strict passes per task: single call vs agent loop: CSV parser",1,"rate","100% (3/3)",3,[0.4385,1],"ci95","Claude Code · single call","\u0001","\u0001","higher","\u0001"],["single-call-vs-agent-loop","agent-loop-by-task","Claude Sonnet 5.5 (single call) · Claude Code","Event-loop order","agent-loop-by-task/Event-loop order","Strict passes per task: single call vs agent loop: Event-loop order",1,"rate","100% (3/3)",3,[0.4385,1],"ci95","Claude Code · single call","\u0001","\u0001","higher","\u0001"],["single-call-vs-agent-loop","agent-loop-by-task","Claude Sonnet 5.5 (single call) · Claude Code","Room schedule","agent-loop-by-task/Room schedule","Strict passes per task: single call vs agent loop: Room schedule",1,"rate","100% (3/3)",3,[0.4385,1],"ci95","Claude Code · single call","\u0001","\u0001","higher","\u0001"],["single-call-vs-agent-loop","agent-loop-by-task","Claude Sonnet 5.5 (single call) · Claude Code","SemVer regex","agent-loop-by-task/SemVer regex","Strict passes per task: single call vs agent loop: SemVer regex",1,"rate","100% (3/3)",3,[0.4385,1],"ci95","Claude Code · single call","\u0001","\u0001","higher","\u0001"],["single-call-vs-agent-loop","agent-loop-by-task","Claude Sonnet 5.5 (single call) · Claude Code","Money refactor","agent-loop-by-task/Money refactor","Strict passes per task: single call vs agent loop: Money refactor",1,"rate","100% (3/3)",3,[0.4385,1],"ci95","Claude Code · single call","\u0001","\u0001","higher","\u0001"],["single-call-vs-agent-loop","agent-loop-by-task","Claude Sonnet 5.5 (single call) · Claude Code","SQL report","agent-loop-by-task/SQL report","Strict passes per task: single call vs agent loop: SQL report",1,"rate","100% (3/3)",3,[0.4385,1],"ci95","Claude Code · single call","\u0001","\u0001","higher","\u0001"],["single-call-vs-agent-loop","agent-loop-by-task","Claude Sonnet 5.5 (agent loop) · Claude Code","Interval merge fix","agent-loop-by-task/Interval merge fix","Strict passes per task: single call vs agent loop: Interval merge fix",1,"rate","100% (2/2)",2,[0.3424,1],"ci95","Claude Code · agent loop","\u0001","\u0001","higher","\u0001"],["single-call-vs-agent-loop","agent-loop-by-task","Claude Sonnet 5.5 (agent loop) · Claude Code","DST day-length fix","agent-loop-by-task/DST day-length fix","Strict passes per task: single call vs agent loop: DST day-length fix",1,"rate","100% (2/2)",2,[0.3424,1],"ci95","Claude Code · agent loop","\u0001","\u0001","higher","\u0001"],["single-call-vs-agent-loop","agent-loop-by-task","Claude Sonnet 5.5 (agent loop) · Claude Code","CSV parser","agent-loop-by-task/CSV parser","Strict passes per task: single call vs agent loop: CSV parser",1,"rate","100% (2/2)",2,[0.3424,1],"ci95","Claude Code · agent loop","\u0001","\u0001","higher","\u0001"],["single-call-vs-agent-loop","agent-loop-by-task","Claude Sonnet 5.5 (agent loop) · Claude Code","Event-loop order","agent-loop-by-task/Event-loop order","Strict passes per task: single call vs agent loop: Event-loop order",1,"rate","100% (2/2)",2,[0.3424,1],"ci95","Claude Code · agent loop","\u0001","\u0001","higher","\u0001"],["single-call-vs-agent-loop","agent-loop-by-task","Claude Sonnet 5.5 (agent loop) · Claude Code","Room schedule","agent-loop-by-task/Room schedule","Strict passes per task: single call vs agent loop: Room schedule",1,"rate","100% (2/2)",2,[0.3424,1],"ci95","Claude Code · agent loop","\u0001","\u0001","higher","\u0001"],["single-call-vs-agent-loop","agent-loop-by-task","Claude Sonnet 5.5 (agent loop) · Claude Code","SemVer regex","agent-loop-by-task/SemVer regex","Strict passes per task: single call vs agent loop: SemVer regex",1,"rate","100% (2/2)",2,[0.3424,1],"ci95","Claude Code · agent loop","\u0001","\u0001","higher","\u0001"],["single-call-vs-agent-loop","agent-loop-by-task","Claude Sonnet 5.5 (agent loop) · Claude Code","Money refactor","agent-loop-by-task/Money refactor","Strict passes per task: single call vs agent loop: Money refactor",1,"rate","100% (2/2)",2,[0.3424,1],"ci95","Claude Code · agent loop","\u0001","\u0001","higher","\u0001"],["single-call-vs-agent-loop","agent-loop-by-task","Claude Sonnet 5.5 (agent loop) · Claude Code","SQL report","agent-loop-by-task/SQL report","Strict passes per task: single call vs agent loop: SQL report",1,"rate","100% (2/2)",2,[0.3424,1],"ci95","Claude Code · agent loop","\u0001","\u0001","higher","\u0001"],["single-call-vs-agent-loop","agent-loop-total-time","Total time per attempt","Claude Sonnet 5.5 (single call) · Claude Code","agent-loop-total-time","Total time per attempt: single call vs agent loop",7.75,"seconds","7.75 s",24,"\u0001","minmax","Claude Code · single call","\u0001",[2.26,34.79],"\u0001","\u0001"],["single-call-vs-agent-loop","agent-loop-total-time","Total time per attempt","Claude Sonnet 5.5 (agent loop) · Claude Code","agent-loop-total-time","Total time per attempt: single call vs agent loop",7.41,"seconds","7.41 s",16,"\u0001","minmax","Claude Code · agent loop","\u0001",[2.75,24.19],"\u0001","\u0001"],["single-call-vs-agent-loop","agent-loop-tokens","Input tokens (cache reads included)","Claude Sonnet 5.5 (single call) · Claude Code","agent-loop-tokens/Input tokens (cache reads included)","Tokens per attempt: single call vs agent loop (Input tokens (cache reads included))",2281,"tokens","2,281",24,"\u0001","minmax","Claude Code · single call","\u0001",[2234,2669],"\u0001","\u0001"],["single-call-vs-agent-loop","agent-loop-tokens","Input tokens (cache reads included)","Claude Sonnet 5.5 (agent loop) · Claude Code","agent-loop-tokens/Input tokens (cache reads included)","Tokens per attempt: single call vs agent loop (Input tokens (cache reads included))",9550,"tokens","9,550",16,"\u0001","minmax","Claude Code · agent loop","\u0001",[9398,33040],"\u0001","\u0001"],["single-call-vs-agent-loop","agent-loop-tokens","Output tokens","Claude Sonnet 5.5 (single call) · Claude Code","agent-loop-tokens/Output tokens","Tokens per attempt: single call vs agent loop (Output tokens)",1050,"tokens","1,050",24,"\u0001","minmax","Claude Code · single call","\u0001",[176,3895],"\u0001","\u0001"],["single-call-vs-agent-loop","agent-loop-tokens","Output tokens","Claude Sonnet 5.5 (agent loop) · Claude Code","agent-loop-tokens/Output tokens","Tokens per attempt: single call vs agent loop (Output tokens)",876,"tokens","876",16,"\u0001","minmax","Claude Code · agent loop","\u0001",[219,3243],"\u0001","\u0001"],["single-call-vs-agent-loop","agent-loop-tool-calls","Tool calls per attempt","Claude Sonnet 5.5 (agent loop) · Claude Code","agent-loop-tool-calls","Tool calls per agent-loop attempt",0,"count","0",16,"\u0001","minmax","Claude Code · agent loop","\u0001",[0,3],"\u0001","\u0001"],["single-call-vs-agent-loop","agent-loop-cost-per-pass","Cost per strict pass","Claude Sonnet 5.5 (single call) · Claude Code","agent-loop-cost-per-pass","List-price cost per strict pass: single call vs agent loop (calculation)",0.01435,"usd","$0.014",24,"\u0001","\u0001","Claude Code · single call",true,"\u0001","\u0001","\u0001"],["single-call-vs-agent-loop","agent-loop-cost-per-pass","Cost per strict pass","Claude Sonnet 5.5 (agent loop) · Claude Code","agent-loop-cost-per-pass","List-price cost per strict pass: single call vs agent loop (calculation)",0.02746,"usd","$0.027",16,"\u0001","\u0001","Claude Code · agent loop",true,"\u0001","\u0001","\u0001"],["haiku-thinking-on-off","haiku-thinking-router-exact","Exact decisions (every scored question right)","Claude Sonnet 5.5 (low) · Claude Code","haiku-thinking-router-exact/Exact decisions (every scored question right)","Haiku thinking study: typed routing decisions answered exactly right (Exact decisions (every scored question right))",0.939,"rate","94% (77/82)",82,[0.8651,0.9737],"ci95","Claude Code · effort low · typed routing decisions, thinking on vs off","\u0001","\u0001","higher","\u0001"],["haiku-thinking-on-off","haiku-thinking-router-exact","Per-question accuracy","Claude Sonnet 5.5 (low) · Claude Code","haiku-thinking-router-exact/Per-question accuracy","Haiku thinking study: typed routing decisions answered exactly right (Per-question accuracy)",0.9742,"rate","97% (189/194)",194,[0.9411,0.9889],"ci95","Claude Code · effort low · typed routing decisions, thinking on vs off","\u0001","\u0001","higher","\u0001"],["haiku-thinking-on-off","haiku-thinking-router-latency","Wall time (CLI)","Claude Sonnet 5.5 (low) · Claude Code","haiku-thinking-router-latency/Wall time (CLI)","Haiku thinking study: time per routing decision (Wall time (CLI))",2.6,"seconds","2.60 s",82,"\u0001","p50-p95","Claude Code · effort low · typed routing decisions, thinking on vs off","\u0001",[2.6,4.3],"\u0001","\u0001"],["haiku-thinking-on-off","haiku-thinking-router-latency","Model time (API)","Claude Sonnet 5.5 (low) · Claude Code","haiku-thinking-router-latency/Model time (API)","Haiku thinking study: time per routing decision (Model time (API))",1.6,"seconds","1.60 s",82,"\u0001","p50-p95","Claude Code · effort low · typed routing decisions, thinking on vs off","\u0001",[1.6,2.58],"\u0001","\u0001"],["haiku-thinking-on-off","haiku-thinking-router-tokens","Thinking tokens","Claude Sonnet 5.5 (low) · Claude Code","haiku-thinking-router-tokens/Thinking tokens","Haiku thinking study: thinking and visible output tokens per routing decision (Thinking tokens)",2,"tokens","2",82,"\u0001","\u0001","Claude Code · effort low · typed routing decisions, thinking on vs off","\u0001","\u0001","\u0001","\u0001"],["haiku-thinking-on-off","haiku-thinking-router-tokens","Visible output tokens","Claude Sonnet 5.5 (low) · Claude Code","haiku-thinking-router-tokens/Visible output tokens","Haiku thinking study: thinking and visible output tokens per routing decision (Visible output tokens)",105,"tokens","105",82,"\u0001","\u0001","Claude Code · effort low · typed routing decisions, thinking on vs off","\u0001","\u0001","\u0001","\u0001"],["haiku-thinking-on-off","haiku-thinking-router-cost","Cost per 1,000 decisions","Claude Sonnet 5.5 (low) · Claude Code","haiku-thinking-router-cost","Haiku thinking study: list-price cost per 1,000 routing decisions (calculation)",7.324,"usd","$7.32",82,"\u0001","\u0001","Claude Code · effort low · typed routing decisions, thinking on vs off",true,"\u0001","\u0001","\u0001"],["json-schema-vs-instructions","structured-output-pass-rate","Strict pass: the whole reply is the right JSON","Claude Sonnet 5.5 (instructions) · Claude Code","structured-output-pass-rate/Strict pass: the whole reply is the right JSON","Does a JSON schema raise the pass rate? Instructions vs schema mode (Strict pass: the whole reply is the right JSON)",1,"rate","100% (12/12)",12,[0.7575,1],"ci95","Claude Code · instructions",true,"\u0001","higher","\u0001"],["json-schema-vs-instructions","structured-output-pass-rate","Strict pass: the whole reply is the right JSON","Claude Sonnet 5.5 (JSON schema) · Claude Code","structured-output-pass-rate/Strict pass: the whole reply is the right JSON","Does a JSON schema raise the pass rate? Instructions vs schema mode (Strict pass: the whole reply is the right JSON)",1,"rate","100% (12/12)",12,[0.7575,1],"ci95","Claude Code · JSON schema",true,"\u0001","higher","\u0001"],["json-schema-vs-instructions","structured-output-pass-rate","Right answer in any format (strict pass or format miss)","Claude Sonnet 5.5 (instructions) · Claude Code","structured-output-pass-rate/Right answer in any format (strict pass or format miss)","Does a JSON schema raise the pass rate? Instructions vs schema mode (Right answer in any format (strict pass or format miss))",1,"rate","100% (12/12)",12,[0.7575,1],"ci95","Claude Code · instructions",true,"\u0001","higher","\u0001"],["json-schema-vs-instructions","structured-output-pass-rate","Right answer in any format (strict pass or format miss)","Claude Sonnet 5.5 (JSON schema) · Claude Code","structured-output-pass-rate/Right answer in any format (strict pass or format miss)","Does a JSON schema raise the pass rate? Instructions vs schema mode (Right answer in any format (strict pass or format miss))",1,"rate","100% (12/12)",12,[0.7575,1],"ci95","Claude Code · JSON schema",true,"\u0001","higher","\u0001"],["json-schema-vs-instructions","structured-output-outcomes","Strict pass","Claude Sonnet 5.5 (instructions) · Claude Code","structured-output-outcomes/Strict pass","What each call produced: strict pass, format miss, wrong values or error (Strict pass)",12,"count","12",12,"\u0001","\u0001","Claude Code · instructions","\u0001","\u0001","\u0001","\u0001"],["json-schema-vs-instructions","structured-output-outcomes","Strict pass","Claude Sonnet 5.5 (JSON schema) · Claude Code","structured-output-outcomes/Strict pass","What each call produced: strict pass, format miss, wrong values or error (Strict pass)",12,"count","12",12,"\u0001","\u0001","Claude Code · JSON schema","\u0001","\u0001","\u0001","\u0001"],["json-schema-vs-instructions","structured-output-outcomes","Format miss","Claude Sonnet 5.5 (instructions) · Claude Code","structured-output-outcomes/Format miss","What each call produced: strict pass, format miss, wrong values or error (Format miss)",0,"count","0",12,"\u0001","\u0001","Claude Code · instructions","\u0001","\u0001","\u0001","\u0001"],["json-schema-vs-instructions","structured-output-outcomes","Format miss","Claude Sonnet 5.5 (JSON schema) · Claude Code","structured-output-outcomes/Format miss","What each call produced: strict pass, format miss, wrong values or error (Format miss)",0,"count","0",12,"\u0001","\u0001","Claude Code · JSON schema","\u0001","\u0001","\u0001","\u0001"],["json-schema-vs-instructions","structured-output-outcomes","Wrong values","Claude Sonnet 5.5 (instructions) · Claude Code","structured-output-outcomes/Wrong values","What each call produced: strict pass, format miss, wrong values or error (Wrong values)",0,"count","0",12,"\u0001","\u0001","Claude Code · instructions","\u0001","\u0001","\u0001","\u0001"],["json-schema-vs-instructions","structured-output-outcomes","Wrong values","Claude Sonnet 5.5 (JSON schema) · Claude Code","structured-output-outcomes/Wrong values","What each call produced: strict pass, format miss, wrong values or error (Wrong values)",0,"count","0",12,"\u0001","\u0001","Claude Code · JSON schema","\u0001","\u0001","\u0001","\u0001"],["json-schema-vs-instructions","structured-output-outcomes","Error","Claude Sonnet 5.5 (instructions) · Claude Code","structured-output-outcomes/Error","What each call produced: strict pass, format miss, wrong values or error (Error)",0,"count","0",12,"\u0001","\u0001","Claude Code · instructions","\u0001","\u0001","\u0001","\u0001"],["json-schema-vs-instructions","structured-output-outcomes","Error","Claude Sonnet 5.5 (JSON schema) · Claude Code","structured-output-outcomes/Error","What each call produced: strict pass, format miss, wrong values or error (Error)",0,"count","0",12,"\u0001","\u0001","Claude Code · JSON schema","\u0001","\u0001","\u0001","\u0001"],["json-schema-vs-instructions","structured-output-time","Median time per call (the three prompts pooled)","Claude Sonnet 5.5 (instructions) · Claude Code","structured-output-time","Time per call, instructions vs schema mode",3.52,"seconds","3.52 s",12,"\u0001","minmax","Claude Code · instructions","\u0001",[2.67,4.12],"\u0001","\u0001"],["json-schema-vs-instructions","structured-output-time","Median time per call (the three prompts pooled)","Claude Sonnet 5.5 (JSON schema) · Claude Code","structured-output-time","Time per call, instructions vs schema mode",4.2,"seconds","4.20 s",12,"\u0001","minmax","Claude Code · JSON schema","\u0001",[2.95,6.14],"\u0001","\u0001"],["json-schema-vs-instructions","structured-output-tokens","Median output tokens per call","Claude Sonnet 5.5 (instructions) · Claude Code","structured-output-tokens/Median output tokens per call","Output and reasoning tokens per call, instructions vs schema mode (Median output tokens per call)",368,"tokens","368",12,"\u0001","\u0001","Claude Code · instructions","\u0001","\u0001","\u0001","\u0001"],["json-schema-vs-instructions","structured-output-tokens","Median output tokens per call","Claude Sonnet 5.5 (JSON schema) · Claude Code","structured-output-tokens/Median output tokens per call","Output and reasoning tokens per call, instructions vs schema mode (Median output tokens per call)",424,"tokens","424",12,"\u0001","\u0001","Claude Code · JSON schema","\u0001","\u0001","\u0001","\u0001"],["prompt-cache-break-even","cache-break-even-reads","1-hour write (2× input), whole prefix new","Claude Sonnet 5.5","cache-break-even-reads/1-hour write (2× input), whole prefix new","Reuses before a cached prefix costs less, by model and write type (calculation) (1-hour write (2× input), whole prefix new)",1.11,"score","1.11","\u0001","\u0001","\u0001","calculation: Agent’s recorded tokens at this model’s list price",true,"\u0001","\u0001","\u0001"],["prompt-cache-break-even","cache-break-even-reads","1-hour write, pooled n = 6 session share, 19% already cached (as recorded)","Claude Sonnet 5.5","cache-break-even-reads/1-hour write, pooled n = 6 session share, 19% already cached (as recorded)","Reuses before a cached prefix costs less, by model and write type (calculation) (1-hour write, pooled n = 6 session share, 19% already cached (as recorded))",0.72,"score","0.72","\u0001","\u0001","\u0001","calculation: Agent’s recorded tokens at this model’s list price",true,"\u0001","\u0001","\u0001"],["prompt-cache-break-even","cache-break-even-reads","5-minute write (1.25× input, an assumption)","Claude Sonnet 5.5","cache-break-even-reads/5-minute write (1.25× input, an assumption)","Reuses before a cached prefix costs less, by model and write type (calculation) (5-minute write (1.25× input, an assumption))",0.28,"score","0.28","\u0001","\u0001","\u0001","calculation: Agent’s recorded tokens at this model’s list price",true,"\u0001","\u0001","\u0001"],["prompt-cache-break-even","cache-break-even-cost-curve","Claude Sonnet 5.5 (no cache)","1 turn","cache-break-even-cost-curve/1 turn","Cost of a reused prefix with and without the cache, by session length (calculation): 1 turn",15.66,"usd","$15.66","\u0001","\u0001","\u0001","no cache",true,"\u0001","\u0001","\u0001"],["prompt-cache-break-even","cache-break-even-cost-curve","Claude Sonnet 5.5 (no cache)","2 turns","cache-break-even-cost-curve/2 turns","Cost of a reused prefix with and without the cache, by session length (calculation): 2 turns",31.32,"usd","$31.32","\u0001","\u0001","\u0001","no cache",true,"\u0001","\u0001","\u0001"],["prompt-cache-break-even","cache-break-even-cost-curve","Claude Sonnet 5.5 (no cache)","3 turns","cache-break-even-cost-curve/3 turns","Cost of a reused prefix with and without the cache, by session length (calculation): 3 turns",46.99,"usd","$46.99","\u0001","\u0001","\u0001","no cache",true,"\u0001","\u0001","\u0001"],["prompt-cache-break-even","cache-break-even-cost-curve","Claude Sonnet 5.5 (no cache)","5 turns","cache-break-even-cost-curve/5 turns","Cost of a reused prefix with and without the cache, by session length (calculation): 5 turns",78.31,"usd","$78.31","\u0001","\u0001","\u0001","no cache",true,"\u0001","\u0001","\u0001"],["prompt-cache-break-even","cache-break-even-cost-curve","Claude Sonnet 5.5 (no cache)","10 turns","cache-break-even-cost-curve/10 turns","Cost of a reused prefix with and without the cache, by session length (calculation): 10 turns",156.62,"usd","$156.62","\u0001","\u0001","\u0001","no cache",true,"\u0001","\u0001","\u0001"],["prompt-cache-break-even","cache-break-even-cost-curve","Claude Sonnet 5.5 (no cache)","20 turns","cache-break-even-cost-curve/20 turns","Cost of a reused prefix with and without the cache, by session length (calculation): 20 turns",313.24,"usd","$313.24","\u0001","\u0001","\u0001","no cache",true,"\u0001","\u0001","\u0001"],["prompt-cache-break-even","cache-break-even-cost-curve","Claude Sonnet 5.5 (1-hour cache write)","1 turn","cache-break-even-cost-curve/1 turn","Cost of a reused prefix with and without the cache, by session length (calculation): 1 turn",31.32,"usd","$31.32","\u0001","\u0001","\u0001","1-hour cache write",true,"\u0001","\u0001","\u0001"],["prompt-cache-break-even","cache-break-even-cost-curve","Claude Sonnet 5.5 (1-hour cache write)","2 turns","cache-break-even-cost-curve/2 turns","Cost of a reused prefix with and without the cache, by session length (calculation): 2 turns",32.89,"usd","$32.89","\u0001","\u0001","\u0001","1-hour cache write",true,"\u0001","\u0001","\u0001"],["prompt-cache-break-even","cache-break-even-cost-curve","Claude Sonnet 5.5 (1-hour cache write)","3 turns","cache-break-even-cost-curve/3 turns","Cost of a reused prefix with and without the cache, by session length (calculation): 3 turns",34.46,"usd","$34.46","\u0001","\u0001","\u0001","1-hour cache write",true,"\u0001","\u0001","\u0001"],["prompt-cache-break-even","cache-break-even-cost-curve","Claude Sonnet 5.5 (1-hour cache write)","5 turns","cache-break-even-cost-curve/5 turns","Cost of a reused prefix with and without the cache, by session length (calculation): 5 turns",37.59,"usd","$37.59","\u0001","\u0001","\u0001","1-hour cache write",true,"\u0001","\u0001","\u0001"],["prompt-cache-break-even","cache-break-even-cost-curve","Claude Sonnet 5.5 (1-hour cache write)","10 turns","cache-break-even-cost-curve/10 turns","Cost of a reused prefix with and without the cache, by session length (calculation): 10 turns",45.42,"usd","$45.42","\u0001","\u0001","\u0001","1-hour cache write",true,"\u0001","\u0001","\u0001"],["prompt-cache-break-even","cache-break-even-cost-curve","Claude Sonnet 5.5 (1-hour cache write)","20 turns","cache-break-even-cost-curve/20 turns","Cost of a reused prefix with and without the cache, by session length (calculation): 20 turns",61.08,"usd","$61.08","\u0001","\u0001","\u0001","1-hour cache write",true,"\u0001","\u0001","\u0001"],["prompt-cache-break-even","cache-break-even-session-split","No cache (the same either way)","Claude Sonnet 5.5","cache-break-even-session-split/No cache (the same either way)","One 10-turn session or ten 1-turn sessions: cost with and without the cache (calculation) (No cache (the same either way))",156.62,"usd","$156.62","\u0001","\u0001","\u0001","",true,"\u0001","\u0001","\u0001"],["prompt-cache-break-even","cache-break-even-session-split","One 10-turn session, 1-hour cache","Claude Sonnet 5.5","cache-break-even-session-split/One 10-turn session, 1-hour cache","One 10-turn session or ten 1-turn sessions: cost with and without the cache (calculation) (One 10-turn session, 1-hour cache)",45.42,"usd","$45.42","\u0001","\u0001","\u0001","",true,"\u0001","\u0001","\u0001"],["prompt-cache-break-even","cache-break-even-session-split","Ten 1-turn sessions, 1-hour cache (assumed no reuse of the new prefix)","Claude Sonnet 5.5","cache-break-even-session-split/Ten 1-turn sessions, 1-hour cache (assumed no reuse of the new prefix)","One 10-turn session or ten 1-turn sessions: cost with and without the cache (calculation) (Ten 1-turn sessions, 1-hour cache (assumed no reuse of the new prefix))",313.24,"usd","$313.24","\u0001","\u0001","\u0001","",true,"\u0001","\u0001","\u0001"],["routing-holdout","routing-holdout-exact","Exact rate","Claude Sonnet 5.5 (low) · Claude Code","routing-holdout-exact","Unseen routing decisions answered exactly right",0.875,"rate","88% (49/56)",56,[0.7637,0.9381],"ci95","Claude Code · effort low","\u0001","\u0001","higher","\u0001"],["routing-holdout","routing-holdout-key-accuracy","Key accuracy","Claude Sonnet 5.5 (low) · Claude Code","routing-holdout-key-accuracy","Per-question accuracy on unseen decisions",0.92,"rate","92% (115/125)",125,[0.859,0.956],"ci95","Claude Code · effort low","\u0001","\u0001","higher","\u0001"],["routing-holdout","routing-holdout-by-purpose","Claude Sonnet 5.5 (low) · Claude Code","Failure class","routing-holdout-by-purpose/Failure class","Exact rate on unseen decisions, by decision type: Failure class",1,"rate","100% (14/14)",14,[0.7847,1],"ci95","Claude Code · effort low","\u0001","\u0001","higher","\u0001"],["routing-holdout","routing-holdout-by-purpose","Claude Sonnet 5.5 (low) · Claude Code","Message intent","routing-holdout-by-purpose/Message intent","Exact rate on unseen decisions, by decision type: Message intent",1,"rate","100% (14/14)",14,[0.7847,1],"ci95","Claude Code · effort low","\u0001","\u0001","higher","\u0001"],["routing-holdout","routing-holdout-by-purpose","Claude Sonnet 5.5 (low) · Claude Code","Is it a rule?","routing-holdout-by-purpose/Is it a rule?","Exact rate on unseen decisions, by decision type: Is it a rule?",0.9286,"rate","93% (13/14)",14,[0.6853,0.9873],"ci95","Claude Code · effort low","\u0001","\u0001","higher","\u0001"],["routing-holdout","routing-holdout-by-purpose","Claude Sonnet 5.5 (low) · Claude Code","Context shape","routing-holdout-by-purpose/Context shape","Exact rate on unseen decisions, by decision type: Context shape",0.5714,"rate","57% (8/14)",14,[0.3259,0.7862],"ci95","Claude Code · effort low","\u0001","\u0001","higher","\u0001"],["routing-holdout","routing-holdout-tuned-vs-unseen","Tuned set (routing-jev-vs-llm)","Claude Sonnet 5.5 (low) · Claude Code","routing-holdout-tuned-vs-unseen/Tuned set (routing-jev-vs-llm)","Tuned case set vs unseen holdout: exact rate per router (Tuned set (routing-jev-vs-llm))",0.939,"rate","94% (77/82)",82,[0.8651,0.9737],"ci95","Claude Code · effort low","\u0001","\u0001","higher","\u0001"],["routing-holdout","routing-holdout-tuned-vs-unseen","Unseen holdout","Claude Sonnet 5.5 (low) · Claude Code","routing-holdout-tuned-vs-unseen/Unseen holdout","Tuned case set vs unseen holdout: exact rate per router (Unseen holdout)",0.875,"rate","88% (49/56)",56,[0.7637,0.9381],"ci95","Claude Code · effort low","\u0001","\u0001","higher","\u0001"],["routing-holdout","routing-holdout-latency","Wall time","Claude Sonnet 5.5 (low) · Claude Code","routing-holdout-latency/Wall time","Time per routing decision, by route (Wall time)",2.359,"seconds","2.36 s",56,"\u0001","p50-p95","Claude Code · effort low","\u0001",[2.359,3.657],"\u0001","\u0001"],["routing-holdout","routing-holdout-latency","Model time (API, CLI-reported)","Claude Sonnet 5.5 (low) · Claude Code","routing-holdout-latency/Model time (API, CLI-reported)","Time per routing decision, by route (Model time (API, CLI-reported))",1.485,"seconds","1.49 s",56,"\u0001","p50-p95","Claude Code · effort low","\u0001",[1.485,2.377],"\u0001","\u0001"],["routing-holdout","routing-holdout-cost-per-1000","Cost","Claude Sonnet 5.5 (low) · Claude Code","routing-holdout-cost-per-1000","Cost per 1,000 unseen routing decisions",7.244,"usd","$7.24",56,"\u0001","\u0001","Claude Code · effort low",true,"\u0001","\u0001","\u0001"],["routing-holdout","\u0001","\u0001","\u0001","stat:holdout-gap-claude-sonnet","Claude Sonnet 5.5 (low) · Claude Code: holdout minus tuned-set exact rate",-0.064,"rate","−6.4 points",56,"\u0001","\u0001","low",true,"\u0001","\u0001","holdout-gap-claude-sonnet"],["thinking-token-bill","thinking-bill-share","Median call","Claude Sonnet 5.5 · Claude Code","thinking-bill-share","Reasoning share of output tokens per call on hard tasks (calculation)",54.54,"percent","54.5%",24,"\u0001","minmax","Claude Code",true,[0,95.91],"none","\u0001"],["thinking-token-bill","thinking-bill-cost-per-call","Reasoning (output tokens)","Claude Sonnet 5.5 · Claude Code","thinking-bill-cost-per-call/Reasoning (output tokens)","List-price cost per call: reasoning, remaining output and input (calculation) (Reasoning (output tokens))",0.006665,"usd","$0.0067",24,"\u0001","\u0001","Claude Code",true,"\u0001","none","\u0001"],["thinking-token-bill","thinking-bill-cost-per-call","Remaining output (visible-answer estimate)","Claude Sonnet 5.5 · Claude Code","thinking-bill-cost-per-call/Remaining output (visible-answer estimate)","List-price cost per call: reasoning, remaining output and input (calculation) (Remaining output (visible-answer estimate))",0.003672,"usd","$0.0037",24,"\u0001","\u0001","Claude Code",true,"\u0001","none","\u0001"],["thinking-token-bill","thinking-bill-cost-per-call","Input (prompt, cache priced)","Claude Sonnet 5.5 · Claude Code","thinking-bill-cost-per-call/Input (prompt, cache priced)","List-price cost per call: reasoning, remaining output and input (calculation) (Input (prompt, cache priced))",0.004012,"usd","$0.0040",24,"\u0001","\u0001","Claude Code",true,"\u0001","none","\u0001"],["thinking-token-bill","thinking-bill-by-effort","Reasoning cost per strict pass","Claude Sonnet 5.5 (low) · Claude Code","thinking-bill-by-effort/Reasoning cost per strict pass","Reasoning cost per strict pass by effort, with the total (calculation) (Reasoning cost per strict pass)",0.004317,"usd","$0.0043",16,"\u0001","\u0001","Claude Code · effort low",true,"\u0001","none","\u0001"],["thinking-token-bill","thinking-bill-by-effort","Reasoning cost per strict pass","Claude Sonnet 5.5 (medium) · Claude Code","thinking-bill-by-effort/Reasoning cost per strict pass","Reasoning cost per strict pass by effort, with the total (calculation) (Reasoning cost per strict pass)",0.005946,"usd","$0.0059",16,"\u0001","\u0001","Claude Code · effort medium",true,"\u0001","none","\u0001"],["thinking-token-bill","thinking-bill-by-effort","Reasoning cost per strict pass","Claude Sonnet 5.5 (high) · Claude Code","thinking-bill-by-effort/Reasoning cost per strict pass","Reasoning cost per strict pass by effort, with the total (calculation) (Reasoning cost per strict pass)",0.009369,"usd","$0.0094",16,"\u0001","\u0001","Claude Code · effort high",true,"\u0001","none","\u0001"],["thinking-token-bill","thinking-bill-by-effort","Reasoning cost per strict pass","Claude Sonnet 5.5 · Claude Code","thinking-bill-by-effort/Reasoning cost per strict pass","Reasoning cost per strict pass by effort, with the total (calculation) (Reasoning cost per strict pass)",0.006299,"usd","$0.0063",16,"\u0001","\u0001","Claude Code",true,"\u0001","none","\u0001"],["thinking-token-bill","thinking-bill-by-effort","Total cost per strict pass","Claude Sonnet 5.5 (low) · Claude Code","thinking-bill-by-effort/Total cost per strict pass","Reasoning cost per strict pass by effort, with the total (calculation) (Total cost per strict pass)",0.012191,"usd","$0.012",16,"\u0001","\u0001","Claude Code · effort low",true,"\u0001","none","\u0001"],["thinking-token-bill","thinking-bill-by-effort","Total cost per strict pass","Claude Sonnet 5.5 (medium) · Claude Code","thinking-bill-by-effort/Total cost per strict pass","Reasoning cost per strict pass by effort, with the total (calculation) (Total cost per strict pass)",0.01352,"usd","$0.014",16,"\u0001","\u0001","Claude Code · effort medium",true,"\u0001","none","\u0001"],["thinking-token-bill","thinking-bill-by-effort","Total cost per strict pass","Claude Sonnet 5.5 (high) · Claude Code","thinking-bill-by-effort/Total cost per strict pass","Reasoning cost per strict pass by effort, with the total (calculation) (Total cost per strict pass)",0.016705,"usd","$0.017",16,"\u0001","\u0001","Claude Code · effort high",true,"\u0001","none","\u0001"],["thinking-token-bill","thinking-bill-by-effort","Total cost per strict pass","Claude Sonnet 5.5 · Claude Code","thinking-bill-by-effort/Total cost per strict pass","Reasoning cost per strict pass by effort, with the total (calculation) (Total cost per strict pass)",0.013978,"usd","$0.014",16,"\u0001","\u0001","Claude Code",true,"\u0001","none","\u0001"],["thinking-token-bill","thinking-bill-short-vs-hard","Eight hard tasks","Claude Sonnet 5.5 · Claude Code","thinking-bill-short-vs-hard/Eight hard tasks","Reasoning share on short tasks vs hard tasks (calculation) (Eight hard tasks)",54.54,"percent","54.5%",24,"\u0001","minmax","Claude Code",true,[0,95.91],"none","\u0001"],["thinking-token-bill","thinking-bill-short-vs-hard","Five short tasks","Claude Sonnet 5.5 · Claude Code","thinking-bill-short-vs-hard/Five short tasks","Reasoning share on short tasks vs hard tasks (calculation) (Five short tasks)",0,"percent","0%",15,"\u0001","minmax","Claude Code",true,[0,72.75],"none","\u0001"],["llm-speed-anatomy","speed-anatomy-first-text","Time to first text","Claude Sonnet 5.5 · Claude Code","speed-anatomy-first-text","Time to first text: a 250-line answer, six models",1.96,"seconds","1.96 s",4,"\u0001","minmax","Claude Code","\u0001",[0.88,4.09],"\u0001","\u0001"],["llm-speed-anatomy","speed-anatomy-output-speed","Visible tokens per second","Claude Sonnet 5.5 · Claude Code","speed-anatomy-output-speed","Output speed after the first text: visible tokens per second (calculation)",231.7,"tokens","232",4,"\u0001","minmax","Claude Code",true,[230.3,233],"\u0001","\u0001"],["llm-speed-anatomy","speed-anatomy-chars-per-second","Characters per second","Claude Sonnet 5.5 · Claude Code","speed-anatomy-chars-per-second","Output speed in characters per second after the first text (calculation)",517,"count","517",4,"\u0001","minmax","Claude Code",true,[513,519],"\u0001","\u0001"],["llm-speed-anatomy","speed-anatomy-prompt-size","Claude Sonnet 5.5 · Claude Code","1k","speed-anatomy-prompt-size/1k","Time to first text as the prompt grows: 1k",1.45,"seconds","1.45 s",3,"\u0001","minmax","Claude Code",true,[1.23,1.72],"\u0001","\u0001"],["llm-speed-anatomy","speed-anatomy-prompt-size","Claude Sonnet 5.5 · Claude Code","16k","speed-anatomy-prompt-size/16k","Time to first text as the prompt grows: 16k",1.78,"seconds","1.78 s",3,"\u0001","minmax","Claude Code",true,[1.64,2.11],"\u0001","\u0001"],["llm-speed-anatomy","speed-anatomy-prompt-size","Claude Sonnet 5.5 · Claude Code","64k","speed-anatomy-prompt-size/64k","Time to first text as the prompt grows: 64k",3.07,"seconds","3.07 s",3,"\u0001","minmax","Claude Code",true,[1.38,3.61],"\u0001","\u0001"],["llm-speed-anatomy","speed-anatomy-total-by-size","1k prompt","Claude Sonnet 5.5 · Claude Code","speed-anatomy-total-by-size/1k prompt","Total time per call by prompt size (1k prompt)",1.78,"seconds","1.78 s",3,"\u0001","minmax","Claude Code","\u0001",[1.57,2.12],"\u0001","\u0001"],["llm-speed-anatomy","speed-anatomy-total-by-size","16k prompt","Claude Sonnet 5.5 · Claude Code","speed-anatomy-total-by-size/16k prompt","Total time per call by prompt size (16k prompt)",2.1,"seconds","2.10 s",3,"\u0001","minmax","Claude Code","\u0001",[1.98,2.48],"\u0001","\u0001"],["llm-speed-anatomy","speed-anatomy-total-by-size","64k prompt","Claude Sonnet 5.5 · Claude Code","speed-anatomy-total-by-size/64k prompt","Total time per call by prompt size (64k prompt)",3.44,"seconds","3.44 s",3,"\u0001","minmax","Claude Code","\u0001",[1.74,4.38],"\u0001","\u0001"],["llm-speed-anatomy","speed-anatomy-lookup-correct","Exact answer","Claude Sonnet 5.5 · Claude Code","speed-anatomy-lookup-correct","Exact lookup answers at the 1k, 16k and 64k prompt-size targets",1,"rate","100% (9/9)",9,[0.7009,1],"ci95","Claude Code","\u0001","\u0001","higher","\u0001"],["haiku-retry-or-escalate","retry-escalate-call-cost-by-task","Claude Sonnet 5.5 · Claude Code","Interval merge fix","retry-escalate-call-cost-by-task/Interval merge fix","List-price cost of one call, Haiku 4.5 vs Sonnet 5.5, on each hard task (calculation): Interval merge fix",0.00557,"usd","$0.0056",3,"\u0001","\u0001","Claude Code",true,"\u0001","\u0001","\u0001"],["haiku-retry-or-escalate","retry-escalate-call-cost-by-task","Claude Sonnet 5.5 · Claude Code","DST day length","retry-escalate-call-cost-by-task/DST day length","List-price cost of one call, Haiku 4.5 vs Sonnet 5.5, on each hard task (calculation): DST day length",0.02532,"usd","$0.025",3,"\u0001","\u0001","Claude Code",true,"\u0001","\u0001","\u0001"],["haiku-retry-or-escalate","retry-escalate-call-cost-by-task","Claude Sonnet 5.5 · Claude Code","CSV parser","retry-escalate-call-cost-by-task/CSV parser","List-price cost of one call, Haiku 4.5 vs Sonnet 5.5, on each hard task (calculation): CSV parser",0.01464,"usd","$0.015",3,"\u0001","\u0001","Claude Code",true,"\u0001","\u0001","\u0001"],["haiku-retry-or-escalate","retry-escalate-call-cost-by-task","Claude Sonnet 5.5 · Claude Code","Event-loop order","retry-escalate-call-cost-by-task/Event-loop order","List-price cost of one call, Haiku 4.5 vs Sonnet 5.5, on each hard task (calculation): Event-loop order",0.01588,"usd","$0.016",3,"\u0001","\u0001","Claude Code",true,"\u0001","\u0001","\u0001"],["haiku-retry-or-escalate","retry-escalate-call-cost-by-task","Claude Sonnet 5.5 · Claude Code","Room schedule","retry-escalate-call-cost-by-task/Room schedule","List-price cost of one call, Haiku 4.5 vs Sonnet 5.5, on each hard task (calculation): Room schedule",0.01243,"usd","$0.012",3,"\u0001","\u0001","Claude Code",true,"\u0001","\u0001","\u0001"],["haiku-retry-or-escalate","retry-escalate-call-cost-by-task","Claude Sonnet 5.5 · Claude Code","SemVer regex","retry-escalate-call-cost-by-task/SemVer regex","List-price cost of one call, Haiku 4.5 vs Sonnet 5.5, on each hard task (calculation): SemVer regex",0.00514,"usd","$0.0051",3,"\u0001","\u0001","Claude Code",true,"\u0001","\u0001","\u0001"],["haiku-retry-or-escalate","retry-escalate-call-cost-by-task","Claude Sonnet 5.5 · Claude Code","Money refactor","retry-escalate-call-cost-by-task/Money refactor","List-price cost of one call, Haiku 4.5 vs Sonnet 5.5, on each hard task (calculation): Money refactor",0.00961,"usd","$0.0096",3,"\u0001","\u0001","Claude Code",true,"\u0001","\u0001","\u0001"],["haiku-retry-or-escalate","retry-escalate-call-cost-by-task","Claude Sonnet 5.5 · Claude Code","SQLite report query","retry-escalate-call-cost-by-task/SQLite report query","List-price cost of one call, Haiku 4.5 vs Sonnet 5.5, on each hard task (calculation): SQLite report query",0.01719,"usd","$0.017",3,"\u0001","\u0001","Claude Code",true,"\u0001","\u0001","\u0001"],["harder-tasks-head-to-head","harder-h2h-pass-rate","Strict pass","Claude Sonnet 5.5 · Claude Code","harder-h2h-pass-rate/Strict pass","Pass rate on 4 harder tasks (Strict pass)",0.375,"rate","38% (6/16)",16,[0.1848,0.6136],"ci95","Claude Code","\u0001","\u0001","higher","\u0001"],["harder-tasks-head-to-head","harder-h2h-pass-rate","Lenient (format misses counted)","Claude Sonnet 5.5 · Claude Code","harder-h2h-pass-rate/Lenient (format misses counted)","Pass rate on 4 harder tasks (Lenient (format misses counted))",0.375,"rate","38% (6/16)",16,[0.1848,0.6136],"ci95","Claude Code","\u0001","\u0001","higher","\u0001"],["harder-tasks-head-to-head","harder-h2h-tool-attempts","Tool attempt","Claude Sonnet 5.5 · Claude Code","harder-h2h-tool-attempts","Calls that tried a tool although tools were off",0.3125,"rate","31% (5/16)",16,[0.1416,0.556],"ci95","Claude Code","\u0001","\u0001","none","\u0001"],["harder-tasks-head-to-head","harder-h2h-pass-by-task","Claude Sonnet 5.5 · Claude Code","10x10 nonogram","harder-h2h-pass-by-task/10x10 nonogram","Strict pass rate by task: 10x10 nonogram",1,"rate","100% (4/4)",4,[0.5101,1],"ci95","Claude Code","\u0001","\u0001","higher","\u0001"],["harder-tasks-head-to-head","harder-h2h-pass-by-task","Claude Sonnet 5.5 · Claude Code","Sudoku, 22 givens","harder-h2h-pass-by-task/Sudoku, 22 givens","Strict pass rate by task: Sudoku, 22 givens",0,"rate","0% (0/4)",4,[0,0.4899],"ci95","Claude Code","\u0001","\u0001","higher","\u0001"],["harder-tasks-head-to-head","harder-h2h-pass-by-task","Claude Sonnet 5.5 · Claude Code","6x6 Skyscrapers","harder-h2h-pass-by-task/6x6 Skyscrapers","Strict pass rate by task: 6x6 Skyscrapers",0,"rate","0% (0/4)",4,[0,0.4899],"ci95","Claude Code","\u0001","\u0001","higher","\u0001"],["harder-tasks-head-to-head","harder-h2h-pass-by-task","Claude Sonnet 5.5 · Claude Code","Seeded shuffle output","harder-h2h-pass-by-task/Seeded shuffle output","Strict pass rate by task: Seeded shuffle output",0.5,"rate","50% (2/4)",4,[0.15,0.85],"ci95","Claude Code","\u0001","\u0001","higher","\u0001"],["harder-tasks-head-to-head","harder-h2h-total-latency","Total time per call","Claude Sonnet 5.5 · Claude Code","harder-h2h-total-latency","Total time per call on harder tasks",70.43,"seconds","70.4 s",12,"\u0001","minmax","Claude Code","\u0001",[4.32,210.08],"\u0001","\u0001"],["harder-tasks-head-to-head","harder-h2h-output-tokens","Output tokens","Claude Sonnet 5.5 · Claude Code","harder-h2h-output-tokens/Output tokens","Output tokens per call on harder tasks (Output tokens)",9287,"tokens","9,287",12,"\u0001","minmax","Claude Code","\u0001",[407,27921],"none","\u0001"],["harder-tasks-head-to-head","harder-h2h-cost-per-pass","Cost per strict pass","Claude Sonnet 5.5 · Claude Code","harder-h2h-cost-per-pass","List-price cost per strict pass on harder tasks (calculation)",0.23843,"usd","$0.24",16,"\u0001","\u0001","Claude Code",true,"\u0001","\u0001","\u0001"]]}