{"$k":["studySlug","chartId","series","point","metric","label","value","unit","display","n","ci","spanKind","context","statId","calculation"],"$r":[["swe-bench-verified","swebench-same-instance-leaderboard","Resolved rate","Agent (Sonnet 5.5, full pipeline)","swebench-same-instance-leaderboard","Resolved rate on the same 33 SWE-bench Verified instances",0.7576,"rate","76% (25/33)",33,[0.5898,0.8717],"ci95","full pipeline on Claude Sonnet 5.5","\u0001","\u0001"],["swe-bench-verified","swebench-by-difficulty-band","Agent","No panel model solved it","swebench-by-difficulty-band/No panel model solved it","Resolved rate by difficulty band: No panel model solved it",0.25,"rate","25% (1/4)",4,[0.0456,0.6994],"ci95","full pipeline on Claude Sonnet 5.5","\u0001","\u0001"],["swe-bench-verified","swebench-by-difficulty-band","Agent","Under half solved it","swebench-by-difficulty-band/Under half solved it","Resolved rate by difficulty band: Under half solved it",0.75,"rate","75% (3/4)",4,[0.3006,0.9544],"ci95","full pipeline on Claude Sonnet 5.5","\u0001","\u0001"],["swe-bench-verified","swebench-by-difficulty-band","Agent","Half or more solved it","swebench-by-difficulty-band/Half or more solved it","Resolved rate by difficulty band: Half or more solved it",0.8182,"rate","82% (9/11)",11,[0.523,0.9486],"ci95","full pipeline on Claude Sonnet 5.5","\u0001","\u0001"],["swe-bench-verified","swebench-by-difficulty-band","Agent","Every panel model solved it","swebench-by-difficulty-band/Every panel model solved it","Resolved rate by difficulty band: Every panel model solved it",0.8571,"rate","86% (12/14)",14,[0.6006,0.9599],"ci95","full pipeline on Claude Sonnet 5.5","\u0001","\u0001"],["swe-bench-verified","swebench-model-calls","Mean calls","Agent","swebench-model-calls","Model calls per instance",49.5,"calls","49.5",33,"\u0001","\u0001","full pipeline on Claude Sonnet 5.5","\u0001","\u0001"],["swe-bench-verified","swebench-views","Agent","Campaign 1: 25-instance sample","swebench-views/Campaign 1: 25-instance sample","Every way to slice the run, with intervals: Campaign 1: 25-instance sample",0.72,"rate","72% (18/25)",25,[0.5242,0.8572],"ci95","full pipeline on Claude Sonnet 5.5","\u0001","\u0001"],["swe-bench-verified","swebench-views","Agent","Campaign 2: 8 compiled-extension instances","swebench-views/Campaign 2: 8 compiled-extension instances","Every way to slice the run, with intervals: Campaign 2: 8 compiled-extension instances",0.875,"rate","88% (7/8)",8,[0.5291,0.9776],"ci95","full pipeline on Claude Sonnet 5.5","\u0001","\u0001"],["swe-bench-verified","swebench-views","Agent","Original seed draw of 25","swebench-views/Original seed draw of 25","Every way to slice the run, with intervals: Original seed draw of 25",0.76,"rate","76% (19/25)",25,[0.5657,0.885],"ci95","full pipeline on Claude Sonnet 5.5","\u0001","\u0001"],["swe-bench-verified","swebench-views","Agent","All 33 attempted","swebench-views/All 33 attempted","Every way to slice the run, with intervals: All 33 attempted",0.7576,"rate","76% (25/33)",33,[0.5898,0.8717],"ci95","full pipeline on Claude Sonnet 5.5","\u0001","\u0001"],["swe-bench-verified","\u0001","\u0001","\u0001","stat:cost-per-attempt","Agent model cost per attempt (notional)",2.81,"usd","$2.81",33,"\u0001","\u0001","full pipeline on Claude Sonnet 5.5","cost-per-attempt",true],["swe-bench-verified","\u0001","\u0001","\u0001","stat:cost-per-resolved","Agent model cost per resolved instance (notional)",3.71,"usd","$3.71",25,"\u0001","\u0001","full pipeline on Claude Sonnet 5.5","cost-per-resolved",true],["swe-bench-verified","\u0001","\u0001","\u0001","stat:median-minutes","Median worker time per attempt",9.6,"minutes","9.6 min",33,"\u0001","\u0001","full pipeline on Claude Sonnet 5.5","median-minutes","\u0001"],["blind-review-head-to-head","\u0001","\u0001","\u0001","stat:ai-preferred-latest","Tasks where the panel preferred the AI change (latest attempt)",0.75,"rate","75% (9/12)",12,[0.4677,0.9111],"ci95","blind panel: Agent change vs merged human change · blind review panel","ai-preferred-latest","\u0001"],["blind-review-head-to-head","\u0001","\u0001","\u0001","stat:ai-preferred-first","Tasks where the panel preferred the AI change (first scored attempt)",0.5,"rate","50% (6/12)",12,[0.2538,0.7462],"ci95","blind panel: Agent change vs merged human change · blind review panel","ai-preferred-first","\u0001"],["blind-review-head-to-head","\u0001","\u0001","\u0001","stat:ai-preferred-all-pairs","All scored pairs where the panel preferred the AI change",0.6,"rate","60% (12/20)",20,[0.3866,0.7812],"ci95","blind panel: Agent change vs merged human change · blind review panel","ai-preferred-all-pairs","\u0001"],["blind-review-head-to-head","\u0001","\u0001","\u0001","stat:ai-preferred-public","Public OSS tasks, latest attempt",0.8,"rate","80% (4/5)",5,[0.3755,0.9638],"ci95","blind panel: Agent change vs merged human change · blind review panel","ai-preferred-public","\u0001"],["blind-review-head-to-head","\u0001","\u0001","\u0001","stat:verdicts-ai","Single critic verdicts that preferred the AI change",0.6894,"rate","69% (91/132)",132,[0.606,0.762],"ci95","blind panel: Agent change vs merged human change · blind review panel","verdicts-ai","\u0001"],["cost-thought-experiments","cost-per-resolved-agent-vs-panel","Cost per resolved instance","Agent (notional)","cost-per-resolved-agent-vs-panel","Recorded cost per resolved instance: Agent vs the public panel",3.706,"usd","$3.71",25,"\u0001","\u0001","full pipeline on Claude Sonnet 5.5 · notional","\u0001",true],["coding-calibration","\u0001","\u0001","\u0001","stat:verified-latest","Verified deliveries, latest build",1,"count","1 of 3",3,"\u0001","\u0001","three real tasks, platform builds compared","verified-latest","\u0001"],["coding-calibration","\u0001","\u0001","\u0001","stat:cost-latest","Notional cost, latest build, all 3 tasks",11.06,"usd","$11.06",3,"\u0001","\u0001","three real tasks, platform builds compared","cost-latest",true],["coding-calibration","\u0001","\u0001","\u0001","stat:refusals-trend","Guardrail refusals, first vs latest slice",19,"count","26 → 19",3,"\u0001","\u0001","three real tasks, platform builds compared","refusals-trend","\u0001"],["coding-calibration","\u0001","\u0001","\u0001","stat:first-run-cost","First calibration run (capped, fastify/session)",4.89,"usd","$4.89, stopped at cap",1,"\u0001","\u0001","three real tasks, platform builds compared","first-run-cost",true]]}