{
  "schema": "agent-public-bench@1",
  "generatedAt": "2026-10-07T00:00:00.000Z",
  "url": "https://agent.sasid.ai/benchmarks/swe-bench-opus-vs-sonnet",
  "study": {
    "slug": "swe-bench-opus-vs-sonnet",
    "title": "Opus 5.5 vs Sonnet 5.5 as the Agent brain on SWE-bench Verified (interim)",
    "seoTitle": "Opus 5.5 vs Sonnet 5.5 on SWE-bench Verified (interim)",
    "description": "Interim: 3 of 8 paired SWE-bench Verified instances graded. Opus 5.5 resolved 2, Sonnet 5.5 1 (McNemar p = 1.0). Cost 2.6×, a list-price calculation.",
    "question": "Does Agent resolve more SWE-bench Verified instances with Claude Opus 5.5 as its brain than with Claude Sonnet 5.5, and at what cost and time?",
    "answer": "Interim, not a final result: 3 of 8 declared pairs are graded. With Claude Opus 5.5 as its brain, Agent resolved 2 of 3 (95% Wilson 21–94%); with Claude Sonnet 5.5 it resolved 1 of 3 (6–79%) on the same instances. The one discordant instance (django__django-10554) went to Opus, and the exact McNemar p is 1.0, so these pairs show no difference. On the same instances Opus cost 2.6× as much, a list-price calculation ($22.76 vs $8.64), and used 1.9× the worker minutes. The arms ran on different platform builds, and the other 5 pairs remain not started in this dataset; the recorded usage-reset date was 2026-10-09.",
    "date": "2026-10-06",
    "updated": "2026-10-06",
    "tags": [
      "swe-bench",
      "claude-opus",
      "claude-sonnet",
      "agent-harness",
      "interim"
    ],
    "method": [
      "A paired probe declared before any Opus run: 8 SWE-bench Verified instances from the 33 that Agent attempted with Sonnet, 2 per difficulty band (1 Sonnet miss and 1 Sonnet resolve each), picked by a fixed rule.",
      "Opus arm: Agent with every model call on Claude Opus 5.5 (routing off), balanced mode, real onboarding, cold start, no cost or time cap, one attempt per instance, on platform build 4f6f4027. Sonnet arm: the earlier campaign attempt on the same instance with the same flags and task spec (builds f0ac3a8a and 236c0d3f).",
      "Grading: Official SWE-bench harness 5.0.2 with the official instance images, amd64 emulation; the gold patch must resolve on this host first. An escalation is graded as delivered; nobody answers it.",
      "A usage gate starts an instance only when the subscription has 3% or more left in both its weekly and 5-hour windows. The run ledger resumes after a halt and never repeats an attempt.",
      "Notional cost: the platform price table at each run commit applied to the recorded tokens. The exact McNemar test reads only the discordant pairs."
    ],
    "caveats": [
      "Interim: 3 of 8 declared pairs. n = 3 supports no ranking: the 95% Wilson intervals overlap almost completely.",
      "Different platform builds: Opus ran on 4f6f4027; the Sonnet attempts ran on f0ac3a8a and 236c0d3f. Platform changes can move results by themselves, so a gap compares Sonnet on an older platform with Opus on the current one, not the models alone.",
      "2 of the Sonnet misses in the declared set (django__django-14631, sympy__sympy-18189) were empty patches from platform holds, not wrong fixes. They are not graded for Opus yet.",
      "The instances were chosen to hold Sonnet misses and resolves in equal numbers, so neither rate estimates SWE-bench Verified as a whole.",
      "Costs are list-price calculations on subscription runs, not invoices. Grading ran under amd64 emulation; contamination is not controlled."
    ],
    "sourceIds": [
      "agent-swebench-opus",
      "agent-swebench-c1",
      "agent-swebench-c2",
      "calc-repricing",
      "price-anthropic"
    ],
    "stats": [
      {
        "id": "swebench-opus-resolved",
        "label": "Claude Opus 5.5 in Agent: resolved, interim (3 of 8 pairs graded)",
        "value": 0.6667,
        "unit": "rate",
        "display": "67% (2/3)",
        "n": 3,
        "ci": [
          0.2077,
          0.9385
        ]
      },
      {
        "id": "swebench-sonnet-resolved",
        "label": "Claude Sonnet 5.5 in Agent: resolved on the same 3 instances",
        "value": 0.3333,
        "unit": "rate",
        "display": "33% (1/3)",
        "n": 3,
        "ci": [
          0.0615,
          0.7923
        ]
      },
      {
        "id": "swebench-opus-sonnet-mcnemar",
        "label": "Exact McNemar p on the graded pairs",
        "value": 1,
        "unit": "score",
        "display": "p = 1.0",
        "n": 3,
        "note": "1 pair where only Opus resolved, 0 pairs where only Sonnet did. The test reads only these; p = 1.0 is no evidence of a difference."
      },
      {
        "id": "swebench-opus-sonnet-graded",
        "label": "Declared pairs graded so far",
        "value": 3,
        "unit": "count",
        "display": "3 of 8",
        "n": 8,
        "note": "5 not started: the usage gate stopped the campaign before django__django-11885. The recorded usage-reset date was 2026-10-09; no resumed attempts are included here. Every started attempt counts."
      },
      {
        "id": "swebench-opus-sonnet-cost-ratio",
        "label": "Opus vs Sonnet list-price cost on the same instances (calculation)",
        "value": 2.63,
        "unit": "ratio",
        "display": "2.6×",
        "n": 3,
        "note": "$22.76 vs $8.64 over the 3 graded instances: notional cost from the platform price table, not an invoice."
      },
      {
        "id": "swebench-opus-sonnet-minutes-ratio",
        "label": "Opus vs Sonnet worker minutes on the same instances (ratio of totals)",
        "value": 1.9,
        "unit": "ratio",
        "display": "1.9×",
        "n": 3,
        "note": "55.3 vs 29.1 worker minutes."
      }
    ],
    "charts": [
      {
        "id": "swebench-opus-sonnet-resolved",
        "title": "Resolved on the same 3 SWE-bench Verified instances (interim)",
        "subtitle": "One attempt per arm per instance, official grader · 95% Wilson intervals",
        "kind": "dot-range",
        "unit": "rate",
        "whisker": "ci95",
        "yLabel": "Resolved",
        "viz": "IntervalDotPlot",
        "series": [
          {
            "name": "Resolved",
            "points": [
              {
                "label": "Claude Opus 5.5 (Agent, new build)",
                "value": 0.6667,
                "lo": 0.2077,
                "hi": 0.9385,
                "n": 3
              },
              {
                "label": "Claude Sonnet 5.5 (Agent, older builds)",
                "value": 0.3333,
                "lo": 0.0615,
                "hi": 0.7923,
                "n": 3
              }
            ]
          }
        ],
        "note": "Interim: 3 of 8 declared pairs are graded; 5 were never started; no resumed attempts are included here. With n = 3 the intervals span most of the axis, so this chart supports no ranking. The instances were chosen to hold Sonnet misses and resolves in equal numbers, so neither rate estimates SWE-bench Verified as a whole.",
        "sourceIds": [
          "agent-swebench-opus",
          "agent-swebench-c1",
          "agent-swebench-c2"
        ]
      },
      {
        "id": "swebench-opus-sonnet-cost-per-attempt",
        "title": "List-price cost per attempt (calculation)",
        "subtitle": "Mean over the same 3 instances; notional cost from the platform price table",
        "kind": "bar",
        "unit": "usd",
        "yLabel": "USD per attempt",
        "viz": "CostBars",
        "series": [
          {
            "name": "List-price cost per attempt",
            "points": [
              {
                "label": "Claude Opus 5.5 (Agent, new build)",
                "value": 7.59,
                "n": 3
              },
              {
                "label": "Claude Sonnet 5.5 (Agent, older builds)",
                "value": 2.88,
                "n": 3
              }
            ]
          }
        ],
        "note": "Calculation, not an invoice: the platform price table at each run commit applied to the recorded tokens of subscription runs (Opus 5.5: $4 input, $20 output, $0.20 cache read per million tokens). Totals $22.76 vs $8.64: 2.6×, a ratio of two calculations.",
        "sourceIds": [
          "agent-swebench-opus",
          "agent-swebench-c1",
          "agent-swebench-c2",
          "calc-repricing",
          "price-anthropic"
        ]
      },
      {
        "id": "swebench-opus-sonnet-cost-by-instance",
        "title": "List-price cost per instance (calculation)",
        "subtitle": "One attempt per arm; notional cost from the platform price table",
        "kind": "grouped-bar",
        "unit": "usd",
        "xLabel": "Instance",
        "yLabel": "USD",
        "viz": "DumbbellPairs",
        "series": [
          {
            "name": "Claude Sonnet 5.5 (Agent, older builds)",
            "points": [
              {
                "label": "django-10554",
                "value": 1.52,
                "n": 1
              },
              {
                "label": "matplotlib-21568",
                "value": 3.02,
                "n": 1
              },
              {
                "label": "django-15022",
                "value": 4.1,
                "n": 1
              }
            ]
          },
          {
            "name": "Claude Opus 5.5 (Agent, new build)",
            "points": [
              {
                "label": "django-10554",
                "value": 9.99,
                "n": 1
              },
              {
                "label": "matplotlib-21568",
                "value": 4.9,
                "n": 1
              },
              {
                "label": "django-15022",
                "value": 7.87,
                "n": 1
              }
            ]
          }
        ],
        "note": "Calculation, not an invoice. Outcomes: django-10554 Opus resolved, Sonnet unresolved; matplotlib-21568 Opus resolved, Sonnet resolved; django-15022 Opus unresolved, Sonnet unresolved.",
        "sourceIds": [
          "agent-swebench-opus",
          "agent-swebench-c1",
          "agent-swebench-c2",
          "calc-repricing",
          "price-anthropic"
        ]
      },
      {
        "id": "swebench-opus-sonnet-minutes",
        "title": "Worker time per attempt",
        "subtitle": "Median minutes; whiskers = fastest and slowest of 3 attempts (not an interval)",
        "kind": "dot-range",
        "unit": "minutes",
        "whisker": "minmax",
        "yLabel": "Minutes",
        "viz": "LatencyLanes",
        "series": [
          {
            "name": "Worker minutes per attempt",
            "points": [
              {
                "label": "Claude Opus 5.5 (Agent, new build)",
                "value": 20.26,
                "lo": 9.76,
                "hi": 25.29,
                "n": 3
              },
              {
                "label": "Claude Sonnet 5.5 (Agent, older builds)",
                "value": 9.37,
                "lo": 4.74,
                "hi": 15,
                "n": 3
              }
            ]
          }
        ],
        "note": "Worker minutes from the run reports. Opus attempts ran one at a time on the newer build; the Sonnet attempts ran in earlier campaigns. 3 attempts per arm is too few to call a difference; a range is not a confidence interval.",
        "sourceIds": [
          "agent-swebench-opus",
          "agent-swebench-c1",
          "agent-swebench-c2"
        ]
      },
      {
        "id": "swebench-opus-sonnet-stage-cost",
        "title": "Where the cost goes: pipeline stages (calculation)",
        "subtitle": "List-price cost per stage, summed over the same 3 instances",
        "kind": "grouped-bar",
        "unit": "usd",
        "xLabel": "Pipeline stage",
        "yLabel": "USD",
        "viz": "DumbbellPairs",
        "series": [
          {
            "name": "Claude Sonnet 5.5 (Agent, older builds)",
            "points": [
              {
                "label": "Research",
                "value": 2.34,
                "n": 3
              },
              {
                "label": "Plan",
                "value": 1.16,
                "n": 3
              },
              {
                "label": "Implement",
                "value": 1.32,
                "n": 3
              },
              {
                "label": "Validate",
                "value": 1.42,
                "n": 3
              },
              {
                "label": "Review",
                "value": 0.51,
                "n": 3
              },
              {
                "label": "Deliver",
                "value": 1.56,
                "n": 3
              }
            ]
          },
          {
            "name": "Claude Opus 5.5 (Agent, new build)",
            "points": [
              {
                "label": "Research",
                "value": 6.28,
                "n": 3
              },
              {
                "label": "Plan",
                "value": 1.11,
                "n": 3
              },
              {
                "label": "Implement",
                "value": 4.94,
                "n": 3
              },
              {
                "label": "Validate",
                "value": 2.97,
                "n": 3
              },
              {
                "label": "Review",
                "value": 2.01,
                "n": 3
              },
              {
                "label": "Deliver",
                "value": 4.34,
                "n": 3
              }
            ]
          }
        ],
        "note": "Calculation, not an invoice: stage costs from the run reports (run-mode stages). Onboarding and calls outside a stage are not in these bars, so the stages sum to less than the attempt totals.",
        "sourceIds": [
          "agent-swebench-opus",
          "agent-swebench-c1",
          "agent-swebench-c2",
          "calc-repricing",
          "price-anthropic"
        ]
      }
    ],
    "tables": [
      {
        "id": "swebench-opus-sonnet-pairs",
        "title": "Same 3 instances, two brains",
        "viz": "PairedOutcomeGrid",
        "columns": [
          {
            "key": "pair",
            "label": "Pair",
            "unit": "text"
          },
          {
            "key": "bothRight",
            "label": "Both resolved",
            "unit": "count"
          },
          {
            "key": "onlyA",
            "label": "Only Opus resolved",
            "unit": "count"
          },
          {
            "key": "onlyB",
            "label": "Only Sonnet resolved",
            "unit": "count"
          },
          {
            "key": "bothWrong",
            "label": "Neither resolved",
            "unit": "count"
          },
          {
            "key": "p",
            "label": "Exact McNemar p"
          }
        ],
        "rows": [
          {
            "pair": "Claude Opus 5.5 vs Claude Sonnet 5.5",
            "bothRight": 1,
            "onlyA": 1,
            "onlyB": 0,
            "bothWrong": 1,
            "p": 1
          }
        ]
      },
      {
        "id": "swebench-opus-sonnet-graded",
        "title": "Graded instances: resolved or not, per brain",
        "viz": "HeatMatrix",
        "columns": [
          {
            "key": "instance",
            "label": "Instance",
            "unit": "text"
          },
          {
            "key": "band",
            "label": "Difficulty band",
            "unit": "text"
          },
          {
            "key": "opus",
            "label": "Claude Opus 5.5 (Agent, new build)",
            "unit": "text"
          },
          {
            "key": "sonnet",
            "label": "Claude Sonnet 5.5 (Agent, older builds)",
            "unit": "text"
          }
        ],
        "rows": [
          {
            "instance": "django__django-10554",
            "band": "No panel model solved it",
            "opus": "resolved",
            "sonnet": "unresolved"
          },
          {
            "instance": "matplotlib__matplotlib-21568",
            "band": "No panel model solved it",
            "opus": "resolved",
            "sonnet": "resolved"
          },
          {
            "instance": "django__django-15022",
            "band": "Under half solved it",
            "opus": "unresolved",
            "sonnet": "unresolved"
          }
        ]
      },
      {
        "id": "swebench-opus-sonnet-instances",
        "title": "All 8 declared instances, in run order (an escalation is graded as delivered)",
        "columns": [
          {
            "key": "instance",
            "label": "Instance",
            "unit": "text"
          },
          {
            "key": "band",
            "label": "Difficulty band",
            "unit": "text"
          },
          {
            "key": "panel",
            "label": "Panel solved (of 11)",
            "unit": "count"
          },
          {
            "key": "opus",
            "label": "Opus 5.5",
            "unit": "text"
          },
          {
            "key": "sonnet",
            "label": "Sonnet 5.5",
            "unit": "text"
          },
          {
            "key": "opusCost",
            "label": "Opus cost (calc.)",
            "unit": "usd"
          },
          {
            "key": "sonnetCost",
            "label": "Sonnet cost (calc.)",
            "unit": "usd"
          },
          {
            "key": "opusMin",
            "label": "Opus min",
            "unit": "minutes"
          },
          {
            "key": "sonnetMin",
            "label": "Sonnet min",
            "unit": "minutes"
          }
        ],
        "rows": [
          {
            "instance": "django__django-10554",
            "band": "No panel model solved it",
            "panel": 0,
            "opus": "resolved",
            "sonnet": "unresolved (escalated)",
            "opusCost": 9.99,
            "sonnetCost": 1.52,
            "opusMin": 25.29,
            "sonnetMin": 4.74
          },
          {
            "instance": "matplotlib__matplotlib-21568",
            "band": "No panel model solved it",
            "panel": 0,
            "opus": "resolved (escalated)",
            "sonnet": "resolved",
            "opusCost": 4.9,
            "sonnetCost": 3.02,
            "opusMin": 9.76,
            "sonnetMin": 9.37
          },
          {
            "instance": "django__django-15022",
            "band": "Under half solved it",
            "panel": 4,
            "opus": "unresolved (escalated)",
            "sonnet": "unresolved",
            "opusCost": 7.87,
            "sonnetCost": 4.1,
            "opusMin": 20.26,
            "sonnetMin": 15
          },
          {
            "instance": "django__django-11885",
            "band": "Under half solved it",
            "panel": 3,
            "opus": "not started",
            "sonnet": "resolved",
            "opusCost": null,
            "sonnetCost": 2.27,
            "opusMin": null,
            "sonnetMin": 7.29
          },
          {
            "instance": "django__django-14631",
            "band": "Half or more solved it",
            "panel": 9,
            "opus": "not started",
            "sonnet": "empty patch (escalated)",
            "opusCost": null,
            "sonnetCost": 2.8,
            "opusMin": null,
            "sonnetMin": 38.17
          },
          {
            "instance": "django__django-12193",
            "band": "Half or more solved it",
            "panel": 10,
            "opus": "not started",
            "sonnet": "resolved",
            "opusCost": null,
            "sonnetCost": 2.06,
            "opusMin": null,
            "sonnetMin": 5.74
          },
          {
            "instance": "sympy__sympy-18189",
            "band": "Every panel model solved it",
            "panel": 11,
            "opus": "not started",
            "sonnet": "empty patch (escalated)",
            "opusCost": null,
            "sonnetCost": 0.89,
            "opusMin": null,
            "sonnetMin": 1.63
          },
          {
            "instance": "django__django-11333",
            "band": "Every panel model solved it",
            "panel": 11,
            "opus": "not started",
            "sonnet": "resolved",
            "opusCost": null,
            "sonnetCost": 2.51,
            "opusMin": null,
            "sonnetMin": 7.99
          }
        ]
      }
    ],
    "related": [
      "swe-bench-verified",
      "coding-agents-head-to-head",
      "effort-ladder"
    ],
    "hero": {
      "statIds": [
        "swebench-opus-resolved",
        "swebench-sonnet-resolved"
      ],
      "testStatId": "swebench-opus-sonnet-mcnemar"
    }
  },
  "sources": [
    {
      "id": "agent-swebench-c1",
      "title": "Agent on SWE-bench Verified, campaign 1 (25 instances)",
      "kind": "run",
      "date": "2026-10-04",
      "note": "Stratified sample of 25 Verified instances (seed 20261004), one attempt each, official grading harness. Fixed model claude-sonnet-5-5, platform build f0ac3a8a.",
      "data": [
        "/benchmarks/raw/swebench/attempts.json",
        "/benchmarks/raw/swebench/exclusions.json"
      ]
    },
    {
      "id": "agent-swebench-c2",
      "title": "Agent on SWE-bench Verified, campaign 2 (8 compiled-extension instances)",
      "kind": "run",
      "date": "2026-10-05",
      "note": "The 6 compiled-extension instances that campaign 1 could not run, plus 2 replacement candidates. One attempt each, platform build 236c0d3f.",
      "data": [
        "/benchmarks/raw/swebench/attempts.json"
      ]
    },
    {
      "id": "agent-swebench-opus",
      "title": "Agent on SWE-bench Verified with Opus 5.5 as the brain, paired with Sonnet 5.5 (interim)",
      "kind": "run",
      "date": "2026-10-06",
      "note": "8 Verified instances declared before the first run, 2 per difficulty band. One Opus attempt each on platform build 4f6f4027, paired with the earlier Sonnet attempt on the same instance; official grader. Interim: a usage gate stopped the campaign after 3 instances; the other 5 resume after the reset on 2026-10-09.",
      "data": [
        "/benchmarks/raw/swebench-opus/instances.json",
        "/benchmarks/raw/swebench/attempts.json"
      ]
    },
    {
      "id": "price-anthropic",
      "title": "Anthropic list prices (Claude models)",
      "kind": "price-list",
      "date": "2026-09-21",
      "url": "https://platform.claude.com/docs/en/about-claude/pricing",
      "note": "Prices as listed by the vendor on 2026-09-21 and recorded in the product price table. Cache reads at the listed rate, one-hour cache writes at twice the input price."
    },
    {
      "id": "calc-repricing",
      "title": "Repricing calculation",
      "kind": "calculation",
      "date": "2026-10-05",
      "note": "Recorded token counts multiplied by the list prices in the price-list sources above. A calculation, not a run: a different model would have used a different number of tokens and reached different outcomes."
    }
  ]
}
