{"$k":["studySlug","chartId","series","point","metric","label","value","unit","display","n","ci","spanKind","context","range","calculation","polarity","statId"],"$r":[["swe-bench-verified","swebench-same-instance-leaderboard","Resolved rate","Claude 4.5 Haiku (high)","swebench-same-instance-leaderboard","Resolved rate on the same 33 SWE-bench Verified instances",0.7576,"rate","76% (25/33)",33,[0.5898,0.8717],"ci95","effort high · public mini-SWE-agent v2 run, same instances","\u0001","\u0001","\u0001","\u0001"],["swe-bench-verified","swebench-model-calls","Mean calls","Claude 4.5 Haiku (high)","swebench-model-calls","Model calls per instance",68.5,"calls","68.5",33,"\u0001","\u0001","effort high · public mini-SWE-agent v2 run, same instances","\u0001","\u0001","\u0001","\u0001"],["model-head-to-head","h2h-pass-rate","Pass rate","Claude Haiku 4.5 · Claude Code","h2h-pass-rate","Pass rate on five validated tasks",1,"rate","100% (15/15)",15,[0.7961,1],"ci95","Claude Code · five short validated tasks","\u0001","\u0001","\u0001","\u0001"],["model-head-to-head","h2h-total-latency","Total time per call","Claude Haiku 4.5 · Claude Code","h2h-total-latency","Total time per call",4.43,"seconds","4.43 s",15,"\u0001","minmax","Claude Code · five short validated tasks",[3.16,23.57],"\u0001","\u0001","\u0001"],["model-head-to-head","h2h-first-useful-latency","Time to first useful output","Claude Haiku 4.5 · Claude Code","h2h-first-useful-latency","Time to first useful output",3.63,"seconds","3.63 s",15,"\u0001","minmax","Claude Code · five short validated tasks",[2.78,22.27],"\u0001","\u0001","\u0001"],["model-head-to-head","h2h-input-tokens","Cache read","Claude Haiku 4.5 · Claude Code","h2h-input-tokens/Cache read","Input tokens per call: what the CLI sends (Cache read)",0,"tokens","0",15,"\u0001","\u0001","Claude Code · five short validated tasks","\u0001","\u0001","\u0001","\u0001"],["model-head-to-head","h2h-input-tokens","Other input","Claude Haiku 4.5 · Claude Code","h2h-input-tokens/Other input","Input tokens per call: what the CLI sends (Other input)",3790,"tokens","3,790",15,"\u0001","\u0001","Claude Code · five short validated tasks","\u0001","\u0001","\u0001","\u0001"],["model-head-to-head","h2h-output-tokens","Output tokens","Claude Haiku 4.5 · Claude Code","h2h-output-tokens/Output tokens","Output tokens per call (Output tokens)",367,"tokens","367",15,"\u0001","\u0001","Claude Code · five short validated tasks","\u0001","\u0001","\u0001","\u0001"],["model-head-to-head","h2h-list-price-per-call","Cost per call","Claude Haiku 4.5 · Claude Code","h2h-list-price-per-call","List-price cost per call (calculation)",0.00566,"usd","$0.0057",15,"\u0001","minmax","Claude Code · five short validated tasks",[0.00513,0.01804],true,"\u0001","\u0001"],["model-head-to-head","h2h-cost-per-pass","Cost per pass","Claude Haiku 4.5 · Claude Code","h2h-cost-per-pass","List-price cost per passing answer (calculation)",0.00836,"usd","$0.0084",15,"\u0001","\u0001","Claude Code · five short validated tasks","\u0001",true,"\u0001","\u0001"],["hard-model-head-to-head","hard-h2h-pass-rate","Strict pass","Claude Haiku 4.5 · Claude Code","hard-h2h-pass-rate/Strict pass","Pass rate on eight hard tasks (Strict pass)",0.4583,"rate","46% (11/24)",24,[0.2789,0.6493],"ci95","Claude Code · eight hard validated tasks","\u0001","\u0001","\u0001","\u0001"],["hard-model-head-to-head","hard-h2h-pass-rate","Lenient (format misses counted)","Claude Haiku 4.5 · Claude Code","hard-h2h-pass-rate/Lenient (format misses counted)","Pass rate on eight hard tasks (Lenient (format misses counted))",0.6667,"rate","67% (16/24)",24,[0.4671,0.8203],"ci95","Claude Code · eight hard validated tasks","\u0001","\u0001","\u0001","\u0001"],["hard-model-head-to-head","hard-h2h-total-latency","Total time per call on hard tasks (separate batches)","Claude Haiku 4.5 · Claude Code","hard-h2h-total-latency","Total time per call on hard tasks (separate batches)",39.01,"seconds","39.0 s",24,"\u0001","minmax","Claude Code · eight hard validated tasks",[15.27,75.13],"\u0001","\u0001","\u0001"],["hard-model-head-to-head","hard-h2h-first-useful-latency","Time to first useful output on hard tasks","Claude Haiku 4.5 · Claude Code","hard-h2h-first-useful-latency","Time to first useful output on hard tasks",35.54,"seconds","35.5 s",24,"\u0001","minmax","Claude Code · eight hard validated tasks",[12.88,70.31],"\u0001","\u0001","\u0001"],["hard-model-head-to-head","hard-h2h-output-tokens","Output tokens","Claude Haiku 4.5 · Claude Code","hard-h2h-output-tokens/Output tokens","Output tokens per call on hard tasks (Output tokens)",5064,"tokens","5,064",24,"\u0001","\u0001","Claude Code · eight hard validated tasks","\u0001","\u0001","\u0001","\u0001"],["hard-model-head-to-head","hard-h2h-cost-per-pass","Cost per strict pass","Claude Haiku 4.5 · Claude Code","hard-h2h-cost-per-pass","List-price cost per strict pass on hard tasks (calculation)",0.0672,"usd","$0.067",24,"\u0001","\u0001","Claude Code · eight hard validated tasks","\u0001",true,"\u0001","\u0001"],["caching-consistency","consistency-pass-rate","Exact number","Claude Haiku 4.5 · Claude Code","consistency-pass-rate/Exact number","Same prompt, 10 times: strict pass rate (Exact number)",0,"rate","0% (0/10)",10,[0,0.2775],"ci95","Claude Code · same prompt repeated 10 times","\u0001","\u0001","\u0001","\u0001"],["caching-consistency","consistency-pass-rate","JSON object","Claude Haiku 4.5 · Claude Code","consistency-pass-rate/JSON object","Same prompt, 10 times: strict pass rate (JSON object)",0.1,"rate","10% (1/10)",10,[0.0179,0.4042],"ci95","Claude Code · same prompt repeated 10 times","\u0001","\u0001","\u0001","\u0001"],["caching-consistency","consistency-pass-rate","Code fix","Claude Haiku 4.5 · Claude Code","consistency-pass-rate/Code fix","Same prompt, 10 times: strict pass rate (Code fix)",1,"rate","100% (10/10)",10,[0.7225,1],"ci95","Claude Code · same prompt repeated 10 times","\u0001","\u0001","\u0001","\u0001"],["caching-consistency","consistency-distinct-answers","Exact number","Claude Haiku 4.5 · Claude Code","consistency-distinct-answers/Exact number","Same prompt, 10 times: how many different answers (Exact number)",1,"count","1",10,"\u0001","\u0001","Claude Code · same prompt repeated 10 times","\u0001","\u0001","\u0001","\u0001"],["caching-consistency","consistency-distinct-answers","JSON object","Claude Haiku 4.5 · Claude Code","consistency-distinct-answers/JSON object","Same prompt, 10 times: how many different answers (JSON object)",1,"count","1",10,"\u0001","\u0001","Claude Code · same prompt repeated 10 times","\u0001","\u0001","\u0001","\u0001"],["caching-consistency","consistency-distinct-answers","Code fix","Claude Haiku 4.5 · Claude Code","consistency-distinct-answers/Code fix","Same prompt, 10 times: how many different answers (Code fix)",6,"count","6",10,"\u0001","\u0001","Claude Code · same prompt repeated 10 times","\u0001","\u0001","\u0001","\u0001"],["caching-consistency","consistency-latency-spread","Exact number","Claude Haiku 4.5 · Claude Code","consistency-latency-spread/Exact number","Same prompt, 10 times: time per call (Exact number)",5.06,"seconds","5.06 s",10,"\u0001","minmax","Claude Code · same prompt repeated 10 times",[4.42,6.2],"\u0001","\u0001","\u0001"],["caching-consistency","consistency-latency-spread","JSON object","Claude Haiku 4.5 · Claude Code","consistency-latency-spread/JSON object","Same prompt, 10 times: time per call (JSON object)",7.03,"seconds","7.03 s",10,"\u0001","minmax","Claude Code · same prompt repeated 10 times",[5.28,12.27],"\u0001","\u0001","\u0001"],["caching-consistency","consistency-latency-spread","Code fix","Claude Haiku 4.5 · Claude Code","consistency-latency-spread/Code fix","Same prompt, 10 times: time per call (Code fix)",5.95,"seconds","5.95 s",10,"\u0001","minmax","Claude Code · same prompt repeated 10 times",[4.89,7.33],"\u0001","\u0001","\u0001"],["agent-memory","memory-full-pass","Claude Haiku 4.5","No memory","memory-full-pass/No memory","Full pass rate by kind of memory: No memory",0.2,"rate","20% (2/10)",10,[0.0567,0.5098],"ci95","","\u0001","\u0001","\u0001","\u0001"],["agent-memory","memory-full-pass","Claude Haiku 4.5","/init CLAUDE.md","memory-full-pass//init CLAUDE.md","Full pass rate by kind of memory: /init CLAUDE.md",0.2,"rate","20% (2/10)",10,[0.0567,0.5098],"ci95","","\u0001","\u0001","\u0001","\u0001"],["agent-memory","memory-full-pass","Claude Haiku 4.5","Curated, 11 lines","memory-full-pass/Curated, 11 lines","Full pass rate by kind of memory: Curated, 11 lines",0.7,"rate","70% (7/10)",10,[0.3968,0.8922],"ci95","","\u0001","\u0001","\u0001","\u0001"],["agent-memory","memory-full-pass","Claude Haiku 4.5","Raw notes, 60 lines","memory-full-pass/Raw notes, 60 lines","Full pass rate by kind of memory: Raw notes, 60 lines",0.6,"rate","60% (6/10)",10,[0.3127,0.8318],"ci95","","\u0001","\u0001","\u0001","\u0001"],["agent-memory","memory-full-pass","Claude Haiku 4.5","Dreamed notes","memory-full-pass/Dreamed notes","Full pass rate by kind of memory: Dreamed notes",0.7,"rate","70% (7/10)",10,[0.3968,0.8922],"ci95","","\u0001","\u0001","\u0001","\u0001"],["agent-memory","memory-full-pass","Claude Haiku 4.5","Handbook, 210 lines","memory-full-pass/Handbook, 210 lines","Full pass rate by kind of memory: Handbook, 210 lines",0.3,"rate","30% (3/10)",10,[0.1078,0.6032],"ci95","","\u0001","\u0001","\u0001","\u0001"],["agent-memory","memory-full-pass","Claude Haiku 4.5","Stop hook only","memory-full-pass/Stop hook only","Full pass rate by kind of memory: Stop hook only",0.8,"rate","80% (8/10)",10,[0.4902,0.9433],"ci95","","\u0001","\u0001","\u0001","\u0001"],["agent-memory","memory-full-pass","Claude Haiku 4.5","Curated + hook","memory-full-pass/Curated + hook","Full pass rate by kind of memory: Curated + hook",0.9,"rate","90% (9/10)",10,[0.5958,0.9821],"ci95","","\u0001","\u0001","\u0001","\u0001"],["agent-memory","memory-team-knowledge-by-model","Claude Haiku 4.5","No memory","memory-team-knowledge-by-model/No memory","Team knowledge followed, Sonnet vs Haiku: No memory",0,"rate","0% (0/10)",10,[0,0.2775],"ci95","","\u0001","\u0001","\u0001","\u0001"],["agent-memory","memory-team-knowledge-by-model","Claude Haiku 4.5","/init CLAUDE.md","memory-team-knowledge-by-model//init CLAUDE.md","Team knowledge followed, Sonnet vs Haiku: /init CLAUDE.md",0.1,"rate","10% (1/10)",10,[0.0179,0.4042],"ci95","","\u0001","\u0001","\u0001","\u0001"],["agent-memory","memory-team-knowledge-by-model","Claude Haiku 4.5","Curated, 11 lines","memory-team-knowledge-by-model/Curated, 11 lines","Team knowledge followed, Sonnet vs Haiku: Curated, 11 lines",0.8,"rate","80% (8/10)",10,[0.4902,0.9433],"ci95","","\u0001","\u0001","\u0001","\u0001"],["agent-memory","memory-team-knowledge-by-model","Claude Haiku 4.5","Raw notes, 60 lines","memory-team-knowledge-by-model/Raw notes, 60 lines","Team knowledge followed, Sonnet vs Haiku: Raw notes, 60 lines",0.6,"rate","60% (6/10)",10,[0.3127,0.8318],"ci95","","\u0001","\u0001","\u0001","\u0001"],["agent-memory","memory-team-knowledge-by-model","Claude Haiku 4.5","Dreamed notes","memory-team-knowledge-by-model/Dreamed notes","Team knowledge followed, Sonnet vs Haiku: Dreamed notes",0.8,"rate","80% (8/10)",10,[0.4902,0.9433],"ci95","","\u0001","\u0001","\u0001","\u0001"],["agent-memory","memory-team-knowledge-by-model","Claude Haiku 4.5","Handbook, 210 lines","memory-team-knowledge-by-model/Handbook, 210 lines","Team knowledge followed, Sonnet vs Haiku: Handbook, 210 lines",0.3,"rate","30% (3/10)",10,[0.1078,0.6032],"ci95","","\u0001","\u0001","\u0001","\u0001"],["agent-memory","memory-team-knowledge-by-model","Claude Haiku 4.5","Stop hook only","memory-team-knowledge-by-model/Stop hook only","Team knowledge followed, Sonnet vs Haiku: Stop hook only",0.8,"rate","80% (8/10)",10,[0.4902,0.9433],"ci95","","\u0001","\u0001","\u0001","\u0001"],["agent-memory","memory-team-knowledge-by-model","Claude Haiku 4.5","Curated + hook","memory-team-knowledge-by-model/Curated + hook","Team knowledge followed, Sonnet vs Haiku: Curated + hook",1,"rate","100% (10/10)",10,[0.7225,1],"ci95","","\u0001","\u0001","\u0001","\u0001"],["agent-memory","memory-broken-test-command","Claude Haiku 4.5","No memory","memory-broken-test-command/No memory","A stale README command: who still ran it?: No memory",1,"rate","100% (10/10)",10,[0.7225,1],"ci95","","\u0001","\u0001","lower","\u0001"],["agent-memory","memory-broken-test-command","Claude Haiku 4.5","/init CLAUDE.md","memory-broken-test-command//init CLAUDE.md","A stale README command: who still ran it?: /init CLAUDE.md",1,"rate","100% (10/10)",10,[0.7225,1],"ci95","","\u0001","\u0001","lower","\u0001"],["agent-memory","memory-broken-test-command","Claude Haiku 4.5","Curated, 11 lines","memory-broken-test-command/Curated, 11 lines","A stale README command: who still ran it?: Curated, 11 lines",0,"rate","0% (0/10)",10,[0,0.2775],"ci95","","\u0001","\u0001","lower","\u0001"],["agent-memory","memory-broken-test-command","Claude Haiku 4.5","Raw notes, 60 lines","memory-broken-test-command/Raw notes, 60 lines","A stale README command: who still ran it?: Raw notes, 60 lines",1,"rate","100% (10/10)",10,[0.7225,1],"ci95","","\u0001","\u0001","lower","\u0001"],["agent-memory","memory-broken-test-command","Claude Haiku 4.5","Dreamed notes","memory-broken-test-command/Dreamed notes","A stale README command: who still ran it?: Dreamed notes",0.1,"rate","10% (1/10)",10,[0.0179,0.4042],"ci95","","\u0001","\u0001","lower","\u0001"],["agent-memory","memory-broken-test-command","Claude Haiku 4.5","Handbook, 210 lines","memory-broken-test-command/Handbook, 210 lines","A stale README command: who still ran it?: Handbook, 210 lines",0,"rate","0% (0/10)",10,[0,0.2775],"ci95","","\u0001","\u0001","lower","\u0001"],["agent-memory","memory-broken-test-command","Claude Haiku 4.5","Stop hook only","memory-broken-test-command/Stop hook only","A stale README command: who still ran it?: Stop hook only",1,"rate","100% (10/10)",10,[0.7225,1],"ci95","","\u0001","\u0001","lower","\u0001"],["agent-memory","memory-broken-test-command","Claude Haiku 4.5","Curated + hook","memory-broken-test-command/Curated + hook","A stale README command: who still ran it?: Curated + hook",0,"rate","0% (0/10)",10,[0,0.2775],"ci95","","\u0001","\u0001","lower","\u0001"],["agent-memory","memory-cost-per-full-pass","Claude Haiku 4.5","No memory","memory-cost-per-full-pass/No memory","List-price cost per fully correct result (calculation): No memory",0.3786,"usd","$0.38",2,"\u0001","\u0001","","\u0001",true,"\u0001","\u0001"],["agent-memory","memory-cost-per-full-pass","Claude Haiku 4.5","/init CLAUDE.md","memory-cost-per-full-pass//init CLAUDE.md","List-price cost per fully correct result (calculation): /init CLAUDE.md",0.4317,"usd","$0.43",2,"\u0001","\u0001","","\u0001",true,"\u0001","\u0001"],["agent-memory","memory-cost-per-full-pass","Claude Haiku 4.5","Curated, 11 lines","memory-cost-per-full-pass/Curated, 11 lines","List-price cost per fully correct result (calculation): Curated, 11 lines",0.1095,"usd","$0.11",7,"\u0001","\u0001","","\u0001",true,"\u0001","\u0001"],["agent-memory","memory-cost-per-full-pass","Claude Haiku 4.5","Raw notes, 60 lines","memory-cost-per-full-pass/Raw notes, 60 lines","List-price cost per fully correct result (calculation): Raw notes, 60 lines",0.1255,"usd","$0.13",6,"\u0001","\u0001","","\u0001",true,"\u0001","\u0001"],["agent-memory","memory-cost-per-full-pass","Claude Haiku 4.5","Dreamed notes","memory-cost-per-full-pass/Dreamed notes","List-price cost per fully correct result (calculation): Dreamed notes",0.1153,"usd","$0.12",7,"\u0001","\u0001","","\u0001",true,"\u0001","\u0001"],["agent-memory","memory-cost-per-full-pass","Claude Haiku 4.5","Handbook, 210 lines","memory-cost-per-full-pass/Handbook, 210 lines","List-price cost per fully correct result (calculation): Handbook, 210 lines",0.2615,"usd","$0.26",3,"\u0001","\u0001","","\u0001",true,"\u0001","\u0001"],["agent-memory","memory-cost-per-full-pass","Claude Haiku 4.5","Stop hook only","memory-cost-per-full-pass/Stop hook only","List-price cost per fully correct result (calculation): Stop hook only",0.1419,"usd","$0.14",8,"\u0001","\u0001","","\u0001",true,"\u0001","\u0001"],["agent-memory","memory-cost-per-full-pass","Claude Haiku 4.5","Curated + hook","memory-cost-per-full-pass/Curated + hook","List-price cost per fully correct result (calculation): Curated + hook",0.096,"usd","$0.096",9,"\u0001","\u0001","","\u0001",true,"\u0001","\u0001"],["agent-memory","memory-wall-time","Claude Haiku 4.5","No memory","memory-wall-time/No memory","Time per session: No memory",54.2,"seconds","54.2 s",10,"\u0001","\u0001","","\u0001","\u0001","\u0001","\u0001"],["agent-memory","memory-wall-time","Claude Haiku 4.5","/init CLAUDE.md","memory-wall-time//init CLAUDE.md","Time per session: /init CLAUDE.md",52.9,"seconds","52.9 s",10,"\u0001","\u0001","","\u0001","\u0001","\u0001","\u0001"],["agent-memory","memory-wall-time","Claude Haiku 4.5","Curated, 11 lines","memory-wall-time/Curated, 11 lines","Time per session: Curated, 11 lines",51.7,"seconds","51.7 s",10,"\u0001","\u0001","","\u0001","\u0001","\u0001","\u0001"],["agent-memory","memory-wall-time","Claude Haiku 4.5","Raw notes, 60 lines","memory-wall-time/Raw notes, 60 lines","Time per session: Raw notes, 60 lines",51,"seconds","51.0 s",10,"\u0001","\u0001","","\u0001","\u0001","\u0001","\u0001"],["agent-memory","memory-wall-time","Claude Haiku 4.5","Dreamed notes","memory-wall-time/Dreamed notes","Time per session: Dreamed notes",51.8,"seconds","51.8 s",10,"\u0001","\u0001","","\u0001","\u0001","\u0001","\u0001"],["agent-memory","memory-wall-time","Claude Haiku 4.5","Handbook, 210 lines","memory-wall-time/Handbook, 210 lines","Time per session: Handbook, 210 lines",49.9,"seconds","49.9 s",10,"\u0001","\u0001","","\u0001","\u0001","\u0001","\u0001"],["agent-memory","memory-wall-time","Claude Haiku 4.5","Stop hook only","memory-wall-time/Stop hook only","Time per session: Stop hook only",68.5,"seconds","68.5 s",10,"\u0001","\u0001","","\u0001","\u0001","\u0001","\u0001"],["agent-memory","memory-wall-time","Claude Haiku 4.5","Curated + hook","memory-wall-time/Curated + hook","Time per session: Curated + hook",52.8,"seconds","52.8 s",10,"\u0001","\u0001","","\u0001","\u0001","\u0001","\u0001"],["routing-jev-vs-llm","routing-exact-decisions","Exact rate","Claude Haiku 4.5","routing-exact-decisions","Typed routing decisions answered exactly right",0.8902,"rate","89% (73/82)",82,[0.8044,0.9412],"ci95","typed routing decisions · via Claude Code","\u0001","\u0001","\u0001","\u0001"],["routing-jev-vs-llm","routing-key-accuracy","Key accuracy","Claude Haiku 4.5","routing-key-accuracy","Per-question accuracy",0.9433,"rate","94% (183/194)",194,[0.9013,0.968],"ci95","typed routing decisions · via Claude Code","\u0001","\u0001","\u0001","\u0001"],["routing-jev-vs-llm","routing-exact-by-decision","Claude Haiku 4.5","Failure class","routing-exact-by-decision/Failure class","Exact rate by decision type: Failure class",0.9444,"rate","94% (17/18)",18,[0.7424,0.9901],"ci95","typed routing decisions · via Claude Code","\u0001","\u0001","\u0001","\u0001"],["routing-jev-vs-llm","routing-exact-by-decision","Claude Haiku 4.5","Message intent","routing-exact-by-decision/Message intent","Exact rate by decision type: Message intent",1,"rate","100% (20/20)",20,[0.8389,1],"ci95","typed routing decisions · via Claude Code","\u0001","\u0001","\u0001","\u0001"],["routing-jev-vs-llm","routing-exact-by-decision","Claude Haiku 4.5","Is it a rule?","routing-exact-by-decision/Is it a rule?","Exact rate by decision type: Is it a rule?",1,"rate","100% (12/12)",12,[0.7575,1],"ci95","typed routing decisions · via Claude Code","\u0001","\u0001","\u0001","\u0001"],["routing-jev-vs-llm","routing-exact-by-decision","Claude Haiku 4.5","Context shape","routing-exact-by-decision/Context shape","Exact rate by decision type: Context shape",0.75,"rate","75% (24/32)",32,[0.5789,0.8675],"ci95","typed routing decisions · via Claude Code","\u0001","\u0001","\u0001","\u0001"],["routing-jev-vs-llm","routing-cost-per-1000","Cost","Claude Haiku 4.5","routing-cost-per-1000","Cost per 1,000 routing decisions",8.924,"usd","$8.92",82,"\u0001","\u0001","typed routing decisions · via Claude Code","\u0001",true,"\u0001","\u0001"],["routing-jev-vs-llm","routing-decision-latency","Wall time (CLI)","Claude Haiku 4.5","routing-decision-latency/Wall time (CLI)","Time per routing decision (Wall time (CLI))",12674,"ms","12,674 ms",82,"\u0001","p50-p95","typed routing decisions · via Claude Code",[12674,34413],"\u0001","\u0001","\u0001"],["routing-jev-vs-llm","routing-decision-latency","Model time (API)","Claude Haiku 4.5","routing-decision-latency/Model time (API)","Time per routing decision (Model time (API))",10734,"ms","10,734 ms",82,"\u0001","p50-p95","typed routing decisions · via Claude Code",[10734,32072],"\u0001","\u0001","\u0001"],["routing-overhead","router-overhead-decision-latency","Decision time","Claude Haiku 4.5 (thinking on, via Claude Code)","router-overhead-decision-latency","Time to make one routing decision",12543,"ms","12,543 ms",82,"\u0001","p50-p95","thinking on · via Claude Code · routing overhead per decision",[12543,34481],"\u0001","\u0001","\u0001"],["routing-overhead","router-overhead-cli-vs-model-time","Model API time","Claude Haiku 4.5 (thinking on, via Claude Code)","router-overhead-cli-vs-model-time/Model API time","Where an LLM router’s time goes: model vs CLI (Model API time)",10508,"ms","10,508 ms",82,"\u0001","p50-p95","thinking on · via Claude Code · routing overhead per decision",[10508,32132],"\u0001","\u0001","\u0001"],["routing-overhead","router-overhead-cli-vs-model-time","CLI and harness time","Claude Haiku 4.5 (thinking on, via Claude Code)","router-overhead-cli-vs-model-time/CLI and harness time","Where an LLM router’s time goes: model vs CLI (CLI and harness time)",1698,"ms","1,698 ms",82,"\u0001","p50-p95","thinking on · via Claude Code · routing overhead per decision",[1698,2677],"\u0001","\u0001","\u0001"],["routing-overhead","router-overhead-completed","Completed","Claude Haiku 4.5 (thinking on, via Claude Code)","router-overhead-completed","Routing calls that returned a decision",1,"rate","100% (82/82)",82,[0.9552,1],"ci95","thinking on · via Claude Code · routing overhead per decision","\u0001","\u0001","\u0001","\u0001"],["routing-overhead","router-overhead-cost-list-price","Cost per 1,000 decisions (list price)","Claude Haiku 4.5 (thinking on, via Claude Code)","router-overhead-cost-list-price","Cost per 1,000 routing decisions for the model routers (calculation)",8.924,"usd","$8.92",82,"\u0001","\u0001","thinking on · via Claude Code · calculation: Agent’s recorded tokens at this model’s list price · routing overhead per decision","\u0001",true,"\u0001","\u0001"],["routing-overhead","router-overhead-cost-per-1000-tasks","Every model call routed (49.5 per task)","Claude Haiku 4.5 (thinking on, via Claude Code)","router-overhead-cost-per-1000-tasks/Every model call routed (49.5 per task)","Added routing cost per 1,000 tasks (calculation) (Every model call routed (49.5 per task))",441.74,"usd","$441.74","\u0001","\u0001","\u0001","thinking on · via Claude Code · calculation per 1,000 tasks from recorded decision counts","\u0001",true,"\u0001","\u0001"],["routing-overhead","router-overhead-cost-per-1000-tasks","Only System One decisions (7 per task)","Claude Haiku 4.5 (thinking on, via Claude Code)","router-overhead-cost-per-1000-tasks/Only System One decisions (7 per task)","Added routing cost per 1,000 tasks (calculation) (Only System One decisions (7 per task))",62.47,"usd","$62.47","\u0001","\u0001","\u0001","thinking on · via Claude Code · calculation per 1,000 tasks from recorded decision counts","\u0001",true,"\u0001","\u0001"],["routing-overhead","router-overhead-delay-per-task","Every model call routed (49.5 per task)","Claude Haiku 4.5 (thinking on, via Claude Code)","router-overhead-delay-per-task/Every model call routed (49.5 per task)","Added routing delay per task (calculation) (Every model call routed (49.5 per task))",620.8785,"seconds","620.9 s","\u0001","\u0001","\u0001","thinking on · via Claude Code · calculation per task from recorded decision counts, decisions in line","\u0001",true,"\u0001","\u0001"],["routing-overhead","router-overhead-delay-per-task","Only System One decisions (7 per task)","Claude Haiku 4.5 (thinking on, via Claude Code)","router-overhead-delay-per-task/Only System One decisions (7 per task)","Added routing delay per task (calculation) (Only System One decisions (7 per task))",87.801,"seconds","87.8 s","\u0001","\u0001","\u0001","thinking on · via Claude Code · calculation per task from recorded decision counts, decisions in line","\u0001",true,"\u0001","\u0001"],["routing-overhead","cli-startup-tax","First output event","Claude Code · Claude Haiku 4.5","cli-startup-tax/First output event","CLI start-up tax on a one-word answer (First output event)",563,"ms","563 ms",5,"\u0001","minmax","Claude Code · CLI start-up, one-word prompt, 5 runs",[519,726],"\u0001","\u0001","\u0001"],["routing-overhead","cli-startup-tax","First model output","Claude Code · Claude Haiku 4.5","cli-startup-tax/First model output","CLI start-up tax on a one-word answer (First model output)",1461,"ms","1,461 ms",5,"\u0001","minmax","Claude Code · CLI start-up, one-word prompt, 5 runs",[1206,2308],"\u0001","\u0001","\u0001"],["routing-overhead","cli-startup-tax","Total wall time","Claude Code · Claude Haiku 4.5","cli-startup-tax/Total wall time","CLI start-up tax on a one-word answer (Total wall time)",2529,"ms","2,529 ms",5,"\u0001","minmax","Claude Code · CLI start-up, one-word prompt, 5 runs",[2273,3382],"\u0001","\u0001","\u0001"],["routing-overhead","cli-startup-input-tokens","Input tokens per call","Claude Code · Claude Haiku 4.5","cli-startup-input-tokens","Input tokens a CLI sends for a one-word answer",6761,"tokens","6,761",5,"\u0001","\u0001","Claude Code · CLI start-up, one-word prompt, 5 runs","\u0001","\u0001","\u0001","\u0001"],["cost-thought-experiments","repriced-cost-per-resolved","Repriced cost per resolved instance","Claude Haiku 4.5","repriced-cost-per-resolved","Thought experiment: the same tokens at other list prices",1.745,"usd","$1.75","\u0001","\u0001","\u0001","calculation: Agent’s recorded tokens at this model’s list price","\u0001",true,"\u0001","\u0001"],["cost-thought-experiments","cost-per-resolved-agent-vs-panel","Cost per resolved instance","Claude 4.5 Haiku (high)","cost-per-resolved-agent-vs-panel","Recorded cost per resolved instance: Agent vs the public panel",0.479,"usd","$0.48",25,"\u0001","\u0001","effort high · public mini-SWE-agent v2 run, same instances","\u0001","\u0001","\u0001","\u0001"],["cost-thought-experiments","prompt-cache-savings","With caching (as recorded)","Claude Haiku 4.5","prompt-cache-savings/With caching (as recorded)","Thought experiment: what prompt caching saved (With caching (as recorded))",43.61,"usd","$43.61","\u0001","\u0001","\u0001","calculation: Agent’s recorded tokens at this model’s list price","\u0001",true,"\u0001","\u0001"],["cost-thought-experiments","prompt-cache-savings","Without caching","Claude Haiku 4.5","prompt-cache-savings/Without caching","Thought experiment: what prompt caching saved (Without caching)",171.66,"usd","$171.66","\u0001","\u0001","\u0001","calculation: Agent’s recorded tokens at this model’s list price","\u0001",true,"\u0001","\u0001"],["single-call-vs-agent-loop","agent-loop-pass-rate","Strict pass","Claude Haiku 4.5 (single call) · Claude Code","agent-loop-pass-rate","Strict pass rate: single call vs agent loop on eight hard tasks",0.4583,"rate","46% (11/24)",24,[0.2789,0.6493],"ci95","Claude Code · single call","\u0001","\u0001","higher","\u0001"],["single-call-vs-agent-loop","agent-loop-pass-rate","Strict pass","Claude Haiku 4.5 (agent loop) · Claude Code","agent-loop-pass-rate","Strict pass rate: single call vs agent loop on eight hard tasks",0.5417,"rate","54% (13/24)",24,[0.3507,0.7211],"ci95","Claude Code · agent loop","\u0001","\u0001","higher","\u0001"],["single-call-vs-agent-loop","agent-loop-by-task","Claude Haiku 4.5 (single call) · Claude Code","Interval merge fix","agent-loop-by-task/Interval merge fix","Strict passes per task: single call vs agent loop: Interval merge fix",1,"rate","100% (3/3)",3,[0.4385,1],"ci95","Claude Code · single call","\u0001","\u0001","higher","\u0001"],["single-call-vs-agent-loop","agent-loop-by-task","Claude Haiku 4.5 (single call) · Claude Code","DST day-length fix","agent-loop-by-task/DST day-length fix","Strict passes per task: single call vs agent loop: DST day-length fix",0.3333,"rate","33% (1/3)",3,[0.0615,0.7923],"ci95","Claude Code · single call","\u0001","\u0001","higher","\u0001"],["single-call-vs-agent-loop","agent-loop-by-task","Claude Haiku 4.5 (single call) · Claude Code","CSV parser","agent-loop-by-task/CSV parser","Strict passes per task: single call vs agent loop: CSV parser",0.6667,"rate","67% (2/3)",3,[0.2077,0.9385],"ci95","Claude Code · single call","\u0001","\u0001","higher","\u0001"],["single-call-vs-agent-loop","agent-loop-by-task","Claude Haiku 4.5 (single call) · Claude Code","Event-loop order","agent-loop-by-task/Event-loop order","Strict passes per task: single call vs agent loop: Event-loop order",0,"rate","0% (0/3)",3,[0,0.5615],"ci95","Claude Code · single call","\u0001","\u0001","higher","\u0001"],["single-call-vs-agent-loop","agent-loop-by-task","Claude Haiku 4.5 (single call) · Claude Code","Room schedule","agent-loop-by-task/Room schedule","Strict passes per task: single call vs agent loop: Room schedule",0,"rate","0% (0/3)",3,[0,0.5615],"ci95","Claude Code · single call","\u0001","\u0001","higher","\u0001"],["single-call-vs-agent-loop","agent-loop-by-task","Claude Haiku 4.5 (single call) · Claude Code","SemVer regex","agent-loop-by-task/SemVer regex","Strict passes per task: single call vs agent loop: SemVer regex",1,"rate","100% (3/3)",3,[0.4385,1],"ci95","Claude Code · single call","\u0001","\u0001","higher","\u0001"],["single-call-vs-agent-loop","agent-loop-by-task","Claude Haiku 4.5 (single call) · Claude Code","Money refactor","agent-loop-by-task/Money refactor","Strict passes per task: single call vs agent loop: Money refactor",0.6667,"rate","67% (2/3)",3,[0.2077,0.9385],"ci95","Claude Code · single call","\u0001","\u0001","higher","\u0001"],["single-call-vs-agent-loop","agent-loop-by-task","Claude Haiku 4.5 (single call) · Claude Code","SQL report","agent-loop-by-task/SQL report","Strict passes per task: single call vs agent loop: SQL report",0,"rate","0% (0/3)",3,[0,0.5615],"ci95","Claude Code · single call","\u0001","\u0001","higher","\u0001"],["single-call-vs-agent-loop","agent-loop-by-task","Claude Haiku 4.5 (agent loop) · Claude Code","Interval merge fix","agent-loop-by-task/Interval merge fix","Strict passes per task: single call vs agent loop: Interval merge fix",0.6667,"rate","67% (2/3)",3,[0.2077,0.9385],"ci95","Claude Code · agent loop","\u0001","\u0001","higher","\u0001"],["single-call-vs-agent-loop","agent-loop-by-task","Claude Haiku 4.5 (agent loop) · Claude Code","DST day-length fix","agent-loop-by-task/DST day-length fix","Strict passes per task: single call vs agent loop: DST day-length fix",0.6667,"rate","67% (2/3)",3,[0.2077,0.9385],"ci95","Claude Code · agent loop","\u0001","\u0001","higher","\u0001"],["single-call-vs-agent-loop","agent-loop-by-task","Claude Haiku 4.5 (agent loop) · Claude Code","CSV parser","agent-loop-by-task/CSV parser","Strict passes per task: single call vs agent loop: CSV parser",0.6667,"rate","67% (2/3)",3,[0.2077,0.9385],"ci95","Claude Code · agent loop","\u0001","\u0001","higher","\u0001"],["single-call-vs-agent-loop","agent-loop-by-task","Claude Haiku 4.5 (agent loop) · Claude Code","Event-loop order","agent-loop-by-task/Event-loop order","Strict passes per task: single call vs agent loop: Event-loop order",1,"rate","100% (3/3)",3,[0.4385,1],"ci95","Claude Code · agent loop","\u0001","\u0001","higher","\u0001"],["single-call-vs-agent-loop","agent-loop-by-task","Claude Haiku 4.5 (agent loop) · Claude Code","Room schedule","agent-loop-by-task/Room schedule","Strict passes per task: single call vs agent loop: Room schedule",0.6667,"rate","67% (2/3)",3,[0.2077,0.9385],"ci95","Claude Code · agent loop","\u0001","\u0001","higher","\u0001"],["single-call-vs-agent-loop","agent-loop-by-task","Claude Haiku 4.5 (agent loop) · Claude Code","SemVer regex","agent-loop-by-task/SemVer regex","Strict passes per task: single call vs agent loop: SemVer regex",0.6667,"rate","67% (2/3)",3,[0.2077,0.9385],"ci95","Claude Code · agent loop","\u0001","\u0001","higher","\u0001"],["single-call-vs-agent-loop","agent-loop-by-task","Claude Haiku 4.5 (agent loop) · Claude Code","Money refactor","agent-loop-by-task/Money refactor","Strict passes per task: single call vs agent loop: Money refactor",0,"rate","0% (0/3)",3,[0,0.5615],"ci95","Claude Code · agent loop","\u0001","\u0001","higher","\u0001"],["single-call-vs-agent-loop","agent-loop-by-task","Claude Haiku 4.5 (agent loop) · Claude Code","SQL report","agent-loop-by-task/SQL report","Strict passes per task: single call vs agent loop: SQL report",0,"rate","0% (0/3)",3,[0,0.5615],"ci95","Claude Code · agent loop","\u0001","\u0001","higher","\u0001"],["single-call-vs-agent-loop","agent-loop-total-time","Total time per attempt","Claude Haiku 4.5 (single call) · Claude Code","agent-loop-total-time","Total time per attempt: single call vs agent loop",39.01,"seconds","39.0 s",24,"\u0001","minmax","Claude Code · single call",[15.27,75.13],"\u0001","\u0001","\u0001"],["single-call-vs-agent-loop","agent-loop-total-time","Total time per attempt","Claude Haiku 4.5 (agent loop) · Claude Code","agent-loop-total-time","Total time per attempt: single call vs agent loop",56.77,"seconds","56.8 s",24,"\u0001","minmax","Claude Code · agent loop",[24.53,223.7],"\u0001","\u0001","\u0001"],["single-call-vs-agent-loop","agent-loop-tokens","Input tokens (cache reads included)","Claude Haiku 4.5 (single call) · Claude Code","agent-loop-tokens/Input tokens (cache reads included)","Tokens per attempt: single call vs agent loop (Input tokens (cache reads included))",3941,"tokens","3,941",24,"\u0001","minmax","Claude Code · single call",[3879,4221],"\u0001","\u0001","\u0001"],["single-call-vs-agent-loop","agent-loop-tokens","Input tokens (cache reads included)","Claude Haiku 4.5 (agent loop) · Claude Code","agent-loop-tokens/Input tokens (cache reads included)","Tokens per attempt: single call vs agent loop (Input tokens (cache reads included))",71691,"tokens","71,691",24,"\u0001","minmax","Claude Code · agent loop",[41732,516306],"\u0001","\u0001","\u0001"],["single-call-vs-agent-loop","agent-loop-tokens","Output tokens","Claude Haiku 4.5 (single call) · Claude Code","agent-loop-tokens/Output tokens","Tokens per attempt: single call vs agent loop (Output tokens)",5064,"tokens","5,064",24,"\u0001","minmax","Claude Code · single call",[1899,9321],"\u0001","\u0001","\u0001"],["single-call-vs-agent-loop","agent-loop-tokens","Output tokens","Claude Haiku 4.5 (agent loop) · Claude Code","agent-loop-tokens/Output tokens","Tokens per attempt: single call vs agent loop (Output tokens)",7912,"tokens","7,912",24,"\u0001","minmax","Claude Code · agent loop",[2541,20654],"\u0001","\u0001","\u0001"],["single-call-vs-agent-loop","agent-loop-tool-calls","Tool calls per attempt","Claude Haiku 4.5 (agent loop) · Claude Code","agent-loop-tool-calls","Tool calls per agent-loop attempt",3,"count","3",24,"\u0001","minmax","Claude Code · agent loop",[2,18],"\u0001","\u0001","\u0001"],["single-call-vs-agent-loop","agent-loop-cost-per-pass","Cost per strict pass","Claude Haiku 4.5 (single call) · Claude Code","agent-loop-cost-per-pass","List-price cost per strict pass: single call vs agent loop (calculation)",0.0672,"usd","$0.067",24,"\u0001","\u0001","Claude Code · single call","\u0001",true,"\u0001","\u0001"],["single-call-vs-agent-loop","agent-loop-cost-per-pass","Cost per strict pass","Claude Haiku 4.5 (agent loop) · Claude Code","agent-loop-cost-per-pass","List-price cost per strict pass: single call vs agent loop (calculation)",0.14225,"usd","$0.14",24,"\u0001","\u0001","Claude Code · agent loop","\u0001",true,"\u0001","\u0001"],["haiku-thinking-on-off","haiku-thinking-router-exact","Exact decisions (every scored question right)","Claude Haiku 4.5 (thinking off) · Claude Code","haiku-thinking-router-exact/Exact decisions (every scored question right)","Haiku thinking study: typed routing decisions answered exactly right (Exact decisions (every scored question right))",0.8659,"rate","87% (71/82)",82,[0.7755,0.9234],"ci95","Claude Code · thinking off · typed routing decisions, thinking on vs off","\u0001","\u0001","higher","\u0001"],["haiku-thinking-on-off","haiku-thinking-router-exact","Exact decisions (every scored question right)","Claude Haiku 4.5 (thinking on) · Claude Code","haiku-thinking-router-exact/Exact decisions (every scored question right)","Haiku thinking study: typed routing decisions answered exactly right (Exact decisions (every scored question right))",0.8902,"rate","89% (73/82)",82,[0.8044,0.9412],"ci95","Claude Code · thinking on · typed routing decisions, thinking on vs off","\u0001","\u0001","higher","\u0001"],["haiku-thinking-on-off","haiku-thinking-router-exact","Per-question accuracy","Claude Haiku 4.5 (thinking off) · Claude Code","haiku-thinking-router-exact/Per-question accuracy","Haiku thinking study: typed routing decisions answered exactly right (Per-question accuracy)",0.9124,"rate","91% (177/194)",194,[0.8642,0.9446],"ci95","Claude Code · thinking off · typed routing decisions, thinking on vs off","\u0001","\u0001","higher","\u0001"],["haiku-thinking-on-off","haiku-thinking-router-exact","Per-question accuracy","Claude Haiku 4.5 (thinking on) · Claude Code","haiku-thinking-router-exact/Per-question accuracy","Haiku thinking study: typed routing decisions answered exactly right (Per-question accuracy)",0.9433,"rate","94% (183/194)",194,[0.9013,0.968],"ci95","Claude Code · thinking on · typed routing decisions, thinking on vs off","\u0001","\u0001","higher","\u0001"],["haiku-thinking-on-off","haiku-thinking-router-latency","Wall time (CLI)","Claude Haiku 4.5 (thinking off) · Claude Code","haiku-thinking-router-latency/Wall time (CLI)","Haiku thinking study: time per routing decision (Wall time (CLI))",4.66,"seconds","4.66 s",82,"\u0001","p50-p95","Claude Code · thinking off · typed routing decisions, thinking on vs off",[4.66,8.18],"\u0001","\u0001","\u0001"],["haiku-thinking-on-off","haiku-thinking-router-latency","Wall time (CLI)","Claude Haiku 4.5 (thinking on) · Claude Code","haiku-thinking-router-latency/Wall time (CLI)","Haiku thinking study: time per routing decision (Wall time (CLI))",12.54,"seconds","12.5 s",82,"\u0001","p50-p95","Claude Code · thinking on · typed routing decisions, thinking on vs off",[12.54,34.48],"\u0001","\u0001","\u0001"],["haiku-thinking-on-off","haiku-thinking-router-latency","Model time (API)","Claude Haiku 4.5 (thinking off) · Claude Code","haiku-thinking-router-latency/Model time (API)","Haiku thinking study: time per routing decision (Model time (API))",3.79,"seconds","3.79 s",82,"\u0001","p50-p95","Claude Code · thinking off · typed routing decisions, thinking on vs off",[3.79,7.43],"\u0001","\u0001","\u0001"],["haiku-thinking-on-off","haiku-thinking-router-latency","Model time (API)","Claude Haiku 4.5 (thinking on) · Claude Code","haiku-thinking-router-latency/Model time (API)","Haiku thinking study: time per routing decision (Model time (API))",10.51,"seconds","10.5 s",82,"\u0001","p50-p95","Claude Code · thinking on · typed routing decisions, thinking on vs off",[10.51,32.13],"\u0001","\u0001","\u0001"],["haiku-thinking-on-off","haiku-thinking-router-tokens","Thinking tokens","Claude Haiku 4.5 (thinking off) · Claude Code","haiku-thinking-router-tokens/Thinking tokens","Haiku thinking study: thinking and visible output tokens per routing decision (Thinking tokens)",0,"tokens","0",82,"\u0001","\u0001","Claude Code · thinking off · typed routing decisions, thinking on vs off","\u0001","\u0001","\u0001","\u0001"],["haiku-thinking-on-off","haiku-thinking-router-tokens","Thinking tokens","Claude Haiku 4.5 (thinking on) · Claude Code","haiku-thinking-router-tokens/Thinking tokens","Haiku thinking study: thinking and visible output tokens per routing decision (Thinking tokens)",1101,"tokens","1,101",82,"\u0001","\u0001","Claude Code · thinking on · typed routing decisions, thinking on vs off","\u0001","\u0001","\u0001","\u0001"],["haiku-thinking-on-off","haiku-thinking-router-tokens","Visible output tokens","Claude Haiku 4.5 (thinking off) · Claude Code","haiku-thinking-router-tokens/Visible output tokens","Haiku thinking study: thinking and visible output tokens per routing decision (Visible output tokens)",366,"tokens","366",82,"\u0001","\u0001","Claude Code · thinking off · typed routing decisions, thinking on vs off","\u0001","\u0001","\u0001","\u0001"],["haiku-thinking-on-off","haiku-thinking-router-tokens","Visible output tokens","Claude Haiku 4.5 (thinking on) · Claude Code","haiku-thinking-router-tokens/Visible output tokens","Haiku thinking study: thinking and visible output tokens per routing decision (Visible output tokens)",318,"tokens","318",82,"\u0001","\u0001","Claude Code · thinking on · typed routing decisions, thinking on vs off","\u0001","\u0001","\u0001","\u0001"],["haiku-thinking-on-off","haiku-thinking-router-cost","Cost per 1,000 decisions","Claude Haiku 4.5 (thinking off) · Claude Code","haiku-thinking-router-cost","Haiku thinking study: list-price cost per 1,000 routing decisions (calculation)",3.364,"usd","$3.36",82,"\u0001","\u0001","Claude Code · thinking off · typed routing decisions, thinking on vs off","\u0001",true,"\u0001","\u0001"],["haiku-thinking-on-off","haiku-thinking-router-cost","Cost per 1,000 decisions","Claude Haiku 4.5 (thinking on) · Claude Code","haiku-thinking-router-cost","Haiku thinking study: list-price cost per 1,000 routing decisions (calculation)",8.924,"usd","$8.92",82,"\u0001","\u0001","Claude Code · thinking on · typed routing decisions, thinking on vs off","\u0001",true,"\u0001","\u0001"],["haiku-thinking-on-off","haiku-thinking-hard-pass","Strict pass","Claude Haiku 4.5 (thinking off) · Claude Code","haiku-thinking-hard-pass/Strict pass","Haiku thinking study: pass rate on eight hard tasks (Strict pass)",0.1667,"rate","17% (4/24)",24,[0.0668,0.3585],"ci95","Claude Code · thinking off · eight hard validated tasks, thinking on vs off","\u0001","\u0001","higher","\u0001"],["haiku-thinking-on-off","haiku-thinking-hard-pass","Strict pass","Claude Haiku 4.5 (thinking on) · Claude Code","haiku-thinking-hard-pass/Strict pass","Haiku thinking study: pass rate on eight hard tasks (Strict pass)",0.4583,"rate","46% (11/24)",24,[0.2789,0.6493],"ci95","Claude Code · thinking on · eight hard validated tasks, thinking on vs off","\u0001","\u0001","higher","\u0001"],["haiku-thinking-on-off","haiku-thinking-hard-pass","Lenient (format misses counted)","Claude Haiku 4.5 (thinking off) · Claude Code","haiku-thinking-hard-pass/Lenient (format misses counted)","Haiku thinking study: pass rate on eight hard tasks (Lenient (format misses counted))",0.1667,"rate","17% (4/24)",24,[0.0668,0.3585],"ci95","Claude Code · thinking off · eight hard validated tasks, thinking on vs off","\u0001","\u0001","higher","\u0001"],["haiku-thinking-on-off","haiku-thinking-hard-pass","Lenient (format misses counted)","Claude Haiku 4.5 (thinking on) · Claude Code","haiku-thinking-hard-pass/Lenient (format misses counted)","Haiku thinking study: pass rate on eight hard tasks (Lenient (format misses counted))",0.6667,"rate","67% (16/24)",24,[0.4671,0.8203],"ci95","Claude Code · thinking on · eight hard validated tasks, thinking on vs off","\u0001","\u0001","higher","\u0001"],["haiku-thinking-on-off","haiku-thinking-hard-time","Total time per call","Claude Haiku 4.5 (thinking off) · Claude Code","haiku-thinking-hard-time","Haiku thinking study: total time per call on hard tasks",2.95,"seconds","2.95 s",24,"\u0001","minmax","Claude Code · thinking off · eight hard validated tasks, thinking on vs off",[1.7,13.01],"\u0001","\u0001","\u0001"],["haiku-thinking-on-off","haiku-thinking-hard-time","Total time per call","Claude Haiku 4.5 (thinking on) · Claude Code","haiku-thinking-hard-time","Haiku thinking study: total time per call on hard tasks",39.01,"seconds","39.0 s",24,"\u0001","minmax","Claude Code · thinking on · eight hard validated tasks, thinking on vs off",[15.27,75.13],"\u0001","\u0001","\u0001"],["haiku-thinking-on-off","\u0001","\u0001","\u0001","stat:haiku-thinking-hard-reasoning-on","Claude Haiku 4.5 (thinking on): median reasoning tokens per hard-task call",4556,"tokens","4,556",24,"\u0001","\u0001","thinking on","\u0001","\u0001","\u0001","haiku-thinking-hard-reasoning-on"],["haiku-thinking-on-off","\u0001","\u0001","\u0001","stat:haiku-thinking-hard-cost-per-pass-off","Claude Haiku 4.5 (thinking off): list-price cost per strict pass on hard tasks (calculation)",0.03654,"usd","$0.0365",24,"\u0001","\u0001","thinking off","\u0001",true,"\u0001","haiku-thinking-hard-cost-per-pass-off"],["haiku-thinking-on-off","\u0001","\u0001","\u0001","stat:haiku-thinking-hard-cost-per-pass-on","Claude Haiku 4.5 (thinking on): list-price cost per strict pass on hard tasks (calculation)",0.0672,"usd","$0.0672",24,"\u0001","\u0001","thinking on","\u0001",true,"\u0001","haiku-thinking-hard-cost-per-pass-on"],["json-schema-vs-instructions","structured-output-pass-rate","Strict pass: the whole reply is the right JSON","Claude Haiku 4.5 (instructions) · Claude Code","structured-output-pass-rate/Strict pass: the whole reply is the right JSON","Does a JSON schema raise the pass rate? Instructions vs schema mode (Strict pass: the whole reply is the right JSON)",0,"rate","0% (0/24)",24,[0,0.138],"ci95","Claude Code · instructions","\u0001",true,"higher","\u0001"],["json-schema-vs-instructions","structured-output-pass-rate","Strict pass: the whole reply is the right JSON","Claude Haiku 4.5 (JSON schema) · Claude Code","structured-output-pass-rate/Strict pass: the whole reply is the right JSON","Does a JSON schema raise the pass rate? Instructions vs schema mode (Strict pass: the whole reply is the right JSON)",0.75,"rate","75% (18/24)",24,[0.551,0.88],"ci95","Claude Code · JSON schema","\u0001",true,"higher","\u0001"],["json-schema-vs-instructions","structured-output-pass-rate","Right answer in any format (strict pass or format miss)","Claude Haiku 4.5 (instructions) · Claude Code","structured-output-pass-rate/Right answer in any format (strict pass or format miss)","Does a JSON schema raise the pass rate? Instructions vs schema mode (Right answer in any format (strict pass or format miss))",0.7083,"rate","71% (17/24)",24,[0.5083,0.8509],"ci95","Claude Code · instructions","\u0001",true,"higher","\u0001"],["json-schema-vs-instructions","structured-output-pass-rate","Right answer in any format (strict pass or format miss)","Claude Haiku 4.5 (JSON schema) · Claude Code","structured-output-pass-rate/Right answer in any format (strict pass or format miss)","Does a JSON schema raise the pass rate? Instructions vs schema mode (Right answer in any format (strict pass or format miss))",0.75,"rate","75% (18/24)",24,[0.551,0.88],"ci95","Claude Code · JSON schema","\u0001",true,"higher","\u0001"],["json-schema-vs-instructions","structured-output-outcomes","Strict pass","Claude Haiku 4.5 (instructions) · Claude Code","structured-output-outcomes/Strict pass","What each call produced: strict pass, format miss, wrong values or error (Strict pass)",0,"count","0",24,"\u0001","\u0001","Claude Code · instructions","\u0001","\u0001","\u0001","\u0001"],["json-schema-vs-instructions","structured-output-outcomes","Strict pass","Claude Haiku 4.5 (JSON schema) · Claude Code","structured-output-outcomes/Strict pass","What each call produced: strict pass, format miss, wrong values or error (Strict pass)",18,"count","18",24,"\u0001","\u0001","Claude Code · JSON schema","\u0001","\u0001","\u0001","\u0001"],["json-schema-vs-instructions","structured-output-outcomes","Format miss","Claude Haiku 4.5 (instructions) · Claude Code","structured-output-outcomes/Format miss","What each call produced: strict pass, format miss, wrong values or error (Format miss)",17,"count","17",24,"\u0001","\u0001","Claude Code · instructions","\u0001","\u0001","\u0001","\u0001"],["json-schema-vs-instructions","structured-output-outcomes","Format miss","Claude Haiku 4.5 (JSON schema) · Claude Code","structured-output-outcomes/Format miss","What each call produced: strict pass, format miss, wrong values or error (Format miss)",0,"count","0",24,"\u0001","\u0001","Claude Code · JSON schema","\u0001","\u0001","\u0001","\u0001"],["json-schema-vs-instructions","structured-output-outcomes","Wrong values","Claude Haiku 4.5 (instructions) · Claude Code","structured-output-outcomes/Wrong values","What each call produced: strict pass, format miss, wrong values or error (Wrong values)",7,"count","7",24,"\u0001","\u0001","Claude Code · instructions","\u0001","\u0001","\u0001","\u0001"],["json-schema-vs-instructions","structured-output-outcomes","Wrong values","Claude Haiku 4.5 (JSON schema) · Claude Code","structured-output-outcomes/Wrong values","What each call produced: strict pass, format miss, wrong values or error (Wrong values)",6,"count","6",24,"\u0001","\u0001","Claude Code · JSON schema","\u0001","\u0001","\u0001","\u0001"],["json-schema-vs-instructions","structured-output-outcomes","Error","Claude Haiku 4.5 (instructions) · Claude Code","structured-output-outcomes/Error","What each call produced: strict pass, format miss, wrong values or error (Error)",0,"count","0",24,"\u0001","\u0001","Claude Code · instructions","\u0001","\u0001","\u0001","\u0001"],["json-schema-vs-instructions","structured-output-outcomes","Error","Claude Haiku 4.5 (JSON schema) · Claude Code","structured-output-outcomes/Error","What each call produced: strict pass, format miss, wrong values or error (Error)",0,"count","0",24,"\u0001","\u0001","Claude Code · JSON schema","\u0001","\u0001","\u0001","\u0001"],["json-schema-vs-instructions","structured-output-time","Median time per call (the three prompts pooled)","Claude Haiku 4.5 (instructions) · Claude Code","structured-output-time","Time per call, instructions vs schema mode",9.52,"seconds","9.52 s",24,"\u0001","minmax","Claude Code · instructions",[5.67,17],"\u0001","\u0001","\u0001"],["json-schema-vs-instructions","structured-output-time","Median time per call (the three prompts pooled)","Claude Haiku 4.5 (JSON schema) · Claude Code","structured-output-time","Time per call, instructions vs schema mode",8.46,"seconds","8.46 s",24,"\u0001","minmax","Claude Code · JSON schema",[5.9,12.23],"\u0001","\u0001","\u0001"],["json-schema-vs-instructions","structured-output-tokens","Median output tokens per call","Claude Haiku 4.5 (instructions) · Claude Code","structured-output-tokens/Median output tokens per call","Output and reasoning tokens per call, instructions vs schema mode (Median output tokens per call)",1128,"tokens","1,128",24,"\u0001","\u0001","Claude Code · instructions","\u0001","\u0001","\u0001","\u0001"],["json-schema-vs-instructions","structured-output-tokens","Median output tokens per call","Claude Haiku 4.5 (JSON schema) · Claude Code","structured-output-tokens/Median output tokens per call","Output and reasoning tokens per call, instructions vs schema mode (Median output tokens per call)",1036,"tokens","1,036",24,"\u0001","\u0001","Claude Code · JSON schema","\u0001","\u0001","\u0001","\u0001"],["prompt-cache-break-even","cache-break-even-reads","1-hour write (2× input), whole prefix new","Claude Haiku 4.5","cache-break-even-reads/1-hour write (2× input), whole prefix new","Reuses before a cached prefix costs less, by model and write type (calculation) (1-hour write (2× input), whole prefix new)",1.11,"score","1.11","\u0001","\u0001","\u0001","calculation: Agent’s recorded tokens at this model’s list price","\u0001",true,"\u0001","\u0001"],["prompt-cache-break-even","cache-break-even-reads","1-hour write, pooled n = 6 session share, 19% already cached (as recorded)","Claude Haiku 4.5","cache-break-even-reads/1-hour write, pooled n = 6 session share, 19% already cached (as recorded)","Reuses before a cached prefix costs less, by model and write type (calculation) (1-hour write, pooled n = 6 session share, 19% already cached (as recorded))",0.72,"score","0.72","\u0001","\u0001","\u0001","calculation: Agent’s recorded tokens at this model’s list price","\u0001",true,"\u0001","\u0001"],["prompt-cache-break-even","cache-break-even-reads","5-minute write (1.25× input, an assumption)","Claude Haiku 4.5","cache-break-even-reads/5-minute write (1.25× input, an assumption)","Reuses before a cached prefix costs less, by model and write type (calculation) (5-minute write (1.25× input, an assumption))",0.28,"score","0.28","\u0001","\u0001","\u0001","calculation: Agent’s recorded tokens at this model’s list price","\u0001",true,"\u0001","\u0001"],["routing-holdout","routing-holdout-exact","Exact rate","Claude Haiku 4.5 · Claude Code","routing-holdout-exact","Unseen routing decisions answered exactly right",0.7857,"rate","79% (44/56)",56,[0.6618,0.8729],"ci95","Claude Code","\u0001","\u0001","higher","\u0001"],["routing-holdout","routing-holdout-key-accuracy","Key accuracy","Claude Haiku 4.5 · Claude Code","routing-holdout-key-accuracy","Per-question accuracy on unseen decisions",0.816,"rate","82% (102/125)",125,[0.739,0.8741],"ci95","Claude Code","\u0001","\u0001","higher","\u0001"],["routing-holdout","routing-holdout-by-purpose","Claude Haiku 4.5 · Claude Code","Failure class","routing-holdout-by-purpose/Failure class","Exact rate on unseen decisions, by decision type: Failure class",0.9286,"rate","93% (13/14)",14,[0.6853,0.9873],"ci95","Claude Code","\u0001","\u0001","higher","\u0001"],["routing-holdout","routing-holdout-by-purpose","Claude Haiku 4.5 · Claude Code","Message intent","routing-holdout-by-purpose/Message intent","Exact rate on unseen decisions, by decision type: Message intent",0.9286,"rate","93% (13/14)",14,[0.6853,0.9873],"ci95","Claude Code","\u0001","\u0001","higher","\u0001"],["routing-holdout","routing-holdout-by-purpose","Claude Haiku 4.5 · Claude Code","Is it a rule?","routing-holdout-by-purpose/Is it a rule?","Exact rate on unseen decisions, by decision type: Is it a rule?",0.9286,"rate","93% (13/14)",14,[0.6853,0.9873],"ci95","Claude Code","\u0001","\u0001","higher","\u0001"],["routing-holdout","routing-holdout-by-purpose","Claude Haiku 4.5 · Claude Code","Context shape","routing-holdout-by-purpose/Context shape","Exact rate on unseen decisions, by decision type: Context shape",0.3571,"rate","36% (5/14)",14,[0.1634,0.6124],"ci95","Claude Code","\u0001","\u0001","higher","\u0001"],["routing-holdout","routing-holdout-tuned-vs-unseen","Tuned set (routing-jev-vs-llm)","Claude Haiku 4.5 · Claude Code","routing-holdout-tuned-vs-unseen/Tuned set (routing-jev-vs-llm)","Tuned case set vs unseen holdout: exact rate per router (Tuned set (routing-jev-vs-llm))",0.8902,"rate","89% (73/82)",82,[0.8044,0.9412],"ci95","Claude Code","\u0001","\u0001","higher","\u0001"],["routing-holdout","routing-holdout-tuned-vs-unseen","Unseen holdout","Claude Haiku 4.5 · Claude Code","routing-holdout-tuned-vs-unseen/Unseen holdout","Tuned case set vs unseen holdout: exact rate per router (Unseen holdout)",0.7857,"rate","79% (44/56)",56,[0.6618,0.8729],"ci95","Claude Code","\u0001","\u0001","higher","\u0001"],["routing-holdout","routing-holdout-latency","Wall time","Claude Haiku 4.5 · Claude Code","routing-holdout-latency/Wall time","Time per routing decision, by route (Wall time)",9.444,"seconds","9.44 s",56,"\u0001","p50-p95","Claude Code",[9.444,25.413],"\u0001","\u0001","\u0001"],["routing-holdout","routing-holdout-latency","Model time (API, CLI-reported)","Claude Haiku 4.5 · Claude Code","routing-holdout-latency/Model time (API, CLI-reported)","Time per routing decision, by route (Model time (API, CLI-reported))",7.522,"seconds","7.52 s",56,"\u0001","p50-p95","Claude Code",[7.522,23.913],"\u0001","\u0001","\u0001"],["routing-holdout","routing-holdout-cost-per-1000","Cost","Claude Haiku 4.5 · Claude Code","routing-holdout-cost-per-1000","Cost per 1,000 unseen routing decisions",7.129,"usd","$7.13",56,"\u0001","\u0001","Claude Code","\u0001",true,"\u0001","\u0001"],["routing-holdout","\u0001","\u0001","\u0001","stat:holdout-gap-claude-haiku","Claude Haiku 4.5 · Claude Code: holdout minus tuned-set exact rate",-0.1045,"rate","−10.5 points",56,"\u0001","\u0001","","\u0001",true,"\u0001","holdout-gap-claude-haiku"],["thinking-token-bill","thinking-bill-share","Median call","Claude Haiku 4.5 · Claude Code","thinking-bill-share","Reasoning share of output tokens per call on hard tasks (calculation)",91.68,"percent","91.7%",24,"\u0001","minmax","Claude Code",[76.46,99.27],true,"none","\u0001"],["thinking-token-bill","thinking-bill-cost-per-call","Reasoning (output tokens)","Claude Haiku 4.5 · Claude Code","thinking-bill-cost-per-call/Reasoning (output tokens)","List-price cost per call: reasoning, remaining output and input (calculation) (Reasoning (output tokens))",0.024492,"usd","$0.024",24,"\u0001","\u0001","Claude Code","\u0001",true,"none","\u0001"],["thinking-token-bill","thinking-bill-cost-per-call","Remaining output (visible-answer estimate)","Claude Haiku 4.5 · Claude Code","thinking-bill-cost-per-call/Remaining output (visible-answer estimate)","List-price cost per call: reasoning, remaining output and input (calculation) (Remaining output (visible-answer estimate))",0.001795,"usd","$0.0018",24,"\u0001","\u0001","Claude Code","\u0001",true,"none","\u0001"],["thinking-token-bill","thinking-bill-cost-per-call","Input (prompt, cache priced)","Claude Haiku 4.5 · Claude Code","thinking-bill-cost-per-call/Input (prompt, cache priced)","List-price cost per call: reasoning, remaining output and input (calculation) (Input (prompt, cache priced))",0.00451,"usd","$0.0045",24,"\u0001","\u0001","Claude Code","\u0001",true,"none","\u0001"],["thinking-token-bill","thinking-bill-short-vs-hard","Eight hard tasks","Claude Haiku 4.5 · Claude Code","thinking-bill-short-vs-hard/Eight hard tasks","Reasoning share on short tasks vs hard tasks (calculation) (Eight hard tasks)",91.68,"percent","91.7%",24,"\u0001","minmax","Claude Code",[76.46,99.27],true,"none","\u0001"],["thinking-token-bill","thinking-bill-short-vs-hard","Five short tasks","Claude Haiku 4.5 · Claude Code","thinking-bill-short-vs-hard/Five short tasks","Reasoning share on short tasks vs hard tasks (calculation) (Five short tasks)",90.19,"percent","90.2%",15,"\u0001","minmax","Claude Code",[73.1,97.59],true,"none","\u0001"],["llm-speed-anatomy","speed-anatomy-first-text","Time to first text","Claude Haiku 4.5 · Claude Code","speed-anatomy-first-text","Time to first text: a 250-line answer, six models",4,"seconds","4.00 s",4,"\u0001","minmax","Claude Code",[2.84,6.38],"\u0001","\u0001","\u0001"],["llm-speed-anatomy","speed-anatomy-output-speed","Visible tokens per second","Claude Haiku 4.5 · Claude Code","speed-anatomy-output-speed","Output speed after the first text: visible tokens per second (calculation)",153.2,"tokens","153",4,"\u0001","minmax","Claude Code",[152.6,216.1],true,"\u0001","\u0001"],["llm-speed-anatomy","speed-anatomy-chars-per-second","Characters per second","Claude Haiku 4.5 · Claude Code","speed-anatomy-chars-per-second","Output speed in characters per second after the first text (calculation)",547,"count","547",3,"\u0001","minmax","Claude Code",[546,548],true,"\u0001","\u0001"],["llm-speed-anatomy","speed-anatomy-prompt-size","Claude Haiku 4.5 · Claude Code","1k","speed-anatomy-prompt-size/1k","Time to first text as the prompt grows: 1k",1.93,"seconds","1.93 s",3,"\u0001","minmax","Claude Code",[1.85,2.04],true,"\u0001","\u0001"],["llm-speed-anatomy","speed-anatomy-prompt-size","Claude Haiku 4.5 · Claude Code","16k","speed-anatomy-prompt-size/16k","Time to first text as the prompt grows: 16k",2.27,"seconds","2.27 s",3,"\u0001","minmax","Claude Code",[2.22,2.47],true,"\u0001","\u0001"],["llm-speed-anatomy","speed-anatomy-prompt-size","Claude Haiku 4.5 · Claude Code","64k","speed-anatomy-prompt-size/64k","Time to first text as the prompt grows: 64k",2.78,"seconds","2.78 s",3,"\u0001","minmax","Claude Code",[2.45,2.89],true,"\u0001","\u0001"],["llm-speed-anatomy","speed-anatomy-total-by-size","1k prompt","Claude Haiku 4.5 · Claude Code","speed-anatomy-total-by-size/1k prompt","Total time per call by prompt size (1k prompt)",2.34,"seconds","2.34 s",3,"\u0001","minmax","Claude Code",[2.22,2.46],"\u0001","\u0001","\u0001"],["llm-speed-anatomy","speed-anatomy-total-by-size","16k prompt","Claude Haiku 4.5 · Claude Code","speed-anatomy-total-by-size/16k prompt","Total time per call by prompt size (16k prompt)",2.79,"seconds","2.79 s",3,"\u0001","minmax","Claude Code",[2.58,2.84],"\u0001","\u0001","\u0001"],["llm-speed-anatomy","speed-anatomy-total-by-size","64k prompt","Claude Haiku 4.5 · Claude Code","speed-anatomy-total-by-size/64k prompt","Total time per call by prompt size (64k prompt)",3.13,"seconds","3.13 s",3,"\u0001","minmax","Claude Code",[2.84,3.28],"\u0001","\u0001","\u0001"],["llm-speed-anatomy","speed-anatomy-lookup-correct","Exact answer","Claude Haiku 4.5 · Claude Code","speed-anatomy-lookup-correct","Exact lookup answers at the 1k, 16k and 64k prompt-size targets",1,"rate","100% (9/9)",9,[0.7009,1],"ci95","Claude Code","\u0001","\u0001","higher","\u0001"],["haiku-retry-or-escalate","retry-escalate-call-cost-by-task","Claude Haiku 4.5 · Claude Code","Interval merge fix","retry-escalate-call-cost-by-task/Interval merge fix","List-price cost of one call, Haiku 4.5 vs Sonnet 5.5, on each hard task (calculation): Interval merge fix",0.01789,"usd","$0.018",3,"\u0001","\u0001","Claude Code","\u0001",true,"\u0001","\u0001"],["haiku-retry-or-escalate","retry-escalate-call-cost-by-task","Claude Haiku 4.5 · Claude Code","DST day length","retry-escalate-call-cost-by-task/DST day length","List-price cost of one call, Haiku 4.5 vs Sonnet 5.5, on each hard task (calculation): DST day length",0.0293,"usd","$0.029",3,"\u0001","\u0001","Claude Code","\u0001",true,"\u0001","\u0001"],["haiku-retry-or-escalate","retry-escalate-call-cost-by-task","Claude Haiku 4.5 · Claude Code","CSV parser","retry-escalate-call-cost-by-task/CSV parser","List-price cost of one call, Haiku 4.5 vs Sonnet 5.5, on each hard task (calculation): CSV parser",0.02919,"usd","$0.029",3,"\u0001","\u0001","Claude Code","\u0001",true,"\u0001","\u0001"],["haiku-retry-or-escalate","retry-escalate-call-cost-by-task","Claude Haiku 4.5 · Claude Code","Event-loop order","retry-escalate-call-cost-by-task/Event-loop order","List-price cost of one call, Haiku 4.5 vs Sonnet 5.5, on each hard task (calculation): Event-loop order",0.03708,"usd","$0.037",3,"\u0001","\u0001","Claude Code","\u0001",true,"\u0001","\u0001"],["haiku-retry-or-escalate","retry-escalate-call-cost-by-task","Claude Haiku 4.5 · Claude Code","Room schedule","retry-escalate-call-cost-by-task/Room schedule","List-price cost of one call, Haiku 4.5 vs Sonnet 5.5, on each hard task (calculation): Room schedule",0.03578,"usd","$0.036",3,"\u0001","\u0001","Claude Code","\u0001",true,"\u0001","\u0001"],["haiku-retry-or-escalate","retry-escalate-call-cost-by-task","Claude Haiku 4.5 · Claude Code","SemVer regex","retry-escalate-call-cost-by-task/SemVer regex","List-price cost of one call, Haiku 4.5 vs Sonnet 5.5, on each hard task (calculation): SemVer regex",0.04217,"usd","$0.042",3,"\u0001","\u0001","Claude Code","\u0001",true,"\u0001","\u0001"],["haiku-retry-or-escalate","retry-escalate-call-cost-by-task","Claude Haiku 4.5 · Claude Code","Money refactor","retry-escalate-call-cost-by-task/Money refactor","List-price cost of one call, Haiku 4.5 vs Sonnet 5.5, on each hard task (calculation): Money refactor",0.02039,"usd","$0.020",3,"\u0001","\u0001","Claude Code","\u0001",true,"\u0001","\u0001"],["haiku-retry-or-escalate","retry-escalate-call-cost-by-task","Claude Haiku 4.5 · Claude Code","SQLite report query","retry-escalate-call-cost-by-task/SQLite report query","List-price cost of one call, Haiku 4.5 vs Sonnet 5.5, on each hard task (calculation): SQLite report query",0.03648,"usd","$0.036",3,"\u0001","\u0001","Claude Code","\u0001",true,"\u0001","\u0001"],["harder-tasks-head-to-head","harder-h2h-pass-rate","Strict pass","Claude Haiku 4.5 · Claude Code","harder-h2h-pass-rate/Strict pass","Pass rate on 4 harder tasks (Strict pass)",0,"rate","0% (0/12)",12,[0,0.2425],"ci95","Claude Code","\u0001","\u0001","higher","\u0001"],["harder-tasks-head-to-head","harder-h2h-pass-rate","Lenient (format misses counted)","Claude Haiku 4.5 · Claude Code","harder-h2h-pass-rate/Lenient (format misses counted)","Pass rate on 4 harder tasks (Lenient (format misses counted))",0,"rate","0% (0/12)",12,[0,0.2425],"ci95","Claude Code","\u0001","\u0001","higher","\u0001"],["harder-tasks-head-to-head","harder-h2h-tool-attempts","Tool attempt","Claude Haiku 4.5 · Claude Code","harder-h2h-tool-attempts","Calls that tried a tool although tools were off",0.0833,"rate","8% (1/12)",12,[0.0149,0.3539],"ci95","Claude Code","\u0001","\u0001","none","\u0001"],["harder-tasks-head-to-head","harder-h2h-pass-by-task","Claude Haiku 4.5 · Claude Code","10x10 nonogram","harder-h2h-pass-by-task/10x10 nonogram","Strict pass rate by task: 10x10 nonogram",0,"rate","0% (0/3)",3,[0,0.5615],"ci95","Claude Code","\u0001","\u0001","higher","\u0001"],["harder-tasks-head-to-head","harder-h2h-pass-by-task","Claude Haiku 4.5 · Claude Code","Sudoku, 22 givens","harder-h2h-pass-by-task/Sudoku, 22 givens","Strict pass rate by task: Sudoku, 22 givens",0,"rate","0% (0/3)",3,[0,0.5615],"ci95","Claude Code","\u0001","\u0001","higher","\u0001"],["harder-tasks-head-to-head","harder-h2h-pass-by-task","Claude Haiku 4.5 · Claude Code","6x6 Skyscrapers","harder-h2h-pass-by-task/6x6 Skyscrapers","Strict pass rate by task: 6x6 Skyscrapers",0,"rate","0% (0/3)",3,[0,0.5615],"ci95","Claude Code","\u0001","\u0001","higher","\u0001"],["harder-tasks-head-to-head","harder-h2h-pass-by-task","Claude Haiku 4.5 · Claude Code","Seeded shuffle output","harder-h2h-pass-by-task/Seeded shuffle output","Strict pass rate by task: Seeded shuffle output",0,"rate","0% (0/3)",3,[0,0.5615],"ci95","Claude Code","\u0001","\u0001","higher","\u0001"],["harder-tasks-head-to-head","harder-h2h-total-latency","Total time per call","Claude Haiku 4.5 · Claude Code","harder-h2h-total-latency","Total time per call on harder tasks",108.98,"seconds","109.0 s",10,"\u0001","minmax","Claude Code",[25.73,223.95],"\u0001","\u0001","\u0001"],["harder-tasks-head-to-head","harder-h2h-output-tokens","Output tokens","Claude Haiku 4.5 · Claude Code","harder-h2h-output-tokens/Output tokens","Output tokens per call on harder tasks (Output tokens)",12508,"tokens","12,508",10,"\u0001","minmax","Claude Code",[2965,26532],"\u0001","none","\u0001"]]}