[{"slug":"blind-review-head-to-head","chart":{"id":"blind-review-critic-agreement","title":"Does the judge’s model family matter?","subtitle":"Share of single verdicts preferring the AI change, per critic model, all scored pairs","kind":"dot-range","unit":"rate","yLabel":"Prefers AI change","series":[{"name":"Critic model","points":{"$k":["label","value","lo","hi","n"],"$r":[["GPT 5.5 (OpenAI)",1,0.3424,1,2],["Codex (GPT) (OpenAI)",0.8,0.4902,0.9433,10],["Claude Opus 5 (Anthropic)",0.7,0.5457,0.8193,40],["Claude Sonnet (Anthropic)",0.675,0.5202,0.7992,40],["Claude Fable 5 (Anthropic)",0.65,0.4951,0.7787,40]]}}],"note":"Whiskers are 95% Wilson intervals over single verdicts (verdicts within one pair are not independent). The worker model is Anthropic; OpenAI critics joined later and judged fewer pairs.","sourceIds":["agent-blind-review"]}},{"slug":"cost-thought-experiments","chart":{"id":"repriced-cost-per-resolved","title":"Thought experiment: the same tokens at other list prices","subtitle":"Cost per resolved SWE-bench instance if 162.9M input and 1.8M output tokens had been billed at each model's list price","kind":"bar","unit":"usd","yLabel":"USD per resolved instance","series":[{"name":"Repriced cost per resolved instance","points":{"$k":["label","value","highlight"],"$r":[["Claude Fable 5.1",12.852,false],["Claude Opus 5",8.723,false],["Claude Opus 5.5",5.753,false],["Claude Sonnet 5.5",3.489,true],["GPT-6.1 Sol",2.097,false],["Claude Haiku 4.5",1.745,false],["Gemini 3.x Flash",1.016,false],["Jev 1.13 (router)",0.042,false]]}}],"note":"Calculation, not a run: tokens recorded by Agent on claude-sonnet-5-5 (33 attempts, 25 resolved) times list prices effective 2026-09-21. Another model would use a different number of tokens and resolve a different set. Jev is a routing model and cannot do this work; its bar is a price floor only.","sourceIds":["calc-repricing","agent-swebench-c1","agent-swebench-c2","price-anthropic","price-google","price-openai","price-jev"]}}]