{
  "schema": "agent-public-bench@1",
  "generatedAt": "2026-10-07T00:00:00.000Z",
  "url": "https://agent.sasid.ai/benchmarks/swe-bench-verified",
  "study": {
    "slug": "swe-bench-verified",
    "title": "Agent on SWE-bench Verified vs 11 public models",
    "seoTitle": "SWE-bench Verified: Agent vs GPT, Claude and Gemini",
    "description": "Agent resolved 25 of 33 SWE-bench Verified instances (76%), inside the public panel's range on the same instances. Cost, time and calls.",
    "question": "How does Agent, a full worker pipeline on one model, do on SWE-bench Verified next to public single-model runs on the very same instances?",
    "answer": "Agent resolved 25 of 33 attempted instances (75.8%, 95% interval 59% to 87%). On the same instances the 11 public mini-SWE-agent v2 runs resolved between 21 and 28 (panel mean 74.1%). Every interval overlaps, so this sample cannot rank Agent above or below any panel model. Agent spent a notional $2.81 and 49 model calls per attempt, with a median of 9.6 minutes; it is slower and more expensive per instance than a bare bash agent because it onboards, plans, verifies and reviews. It resolved 1 of 4 instances that no panel model solved.",
    "date": "2026-10-05",
    "updated": "2026-10-05",
    "tags": [
      "swe-bench",
      "coding-agents",
      "leaderboard",
      "cost",
      "claude-sonnet"
    ],
    "method": [
      "Sample: 25 of the 500 Verified instances, stratified by public difficulty (seed 20261004). Difficulty is how many of the 11 public mini-SWE-agent v2 runs solved the instance.",
      "Campaign 1 ran 25 instances on platform build f0ac3a8a. Six compiled-extension instances could not import in the worker checkout, so the declared rule replaced them in the same band.",
      "Campaign 2 ran those compiled instances (plus two replacement candidates) on build 236c0d3f after a sandbox fix.",
      "One attempt per instance. No retries, no operator answers: an escalation is graded on what was delivered.",
      "Grading uses the official SWE-bench harness and instance images (under amd64 emulation). The gold patch resolved on the host for every instance.",
      "Agent runs its full pipeline (onboarding, research, plan, act, verify, review) on claude-sonnet-5-5 through a subscription CLI. Costs are list-price estimates of the recorded tokens."
    ],
    "caveats": [
      "n = 33: intervals are wide. This is a defect-finding run, not a ranking.",
      "Systems, not models: the panel is one model in a bash-only harness; Agent is a full pipeline on one model.",
      "Panel costs are published API costs; Agent costs are notional subscription estimates, not invoices.",
      "Campaign 1 ended up 19 django, 3 sympy, 2 sphinx and 1 xarray after replacements. Campaign 2 covers the compiled repositories.",
      "Verified issues are public (2015 to 2023) and likely in every model’s training data; contamination is uncontrolled for all systems.",
      "Difficulty bands come from the panel’s own results, so a system outside the panel tends to look better than the panel on hard bands and worse on easy ones (regression to the mean). Read the band chart with that selection effect in mind.",
      "Three campaign-1 empty patches were platform holds before delivery (missing lint tools, a too-literal plan gate, an unanswered question), not wrong fixes. They count as failures here."
    ],
    "sourceIds": [
      "agent-swebench-c1",
      "agent-swebench-c2",
      "swebench-leaderboard",
      "swebench-protocol",
      "price-anthropic"
    ],
    "stats": [
      {
        "id": "agent-rate-33",
        "label": "Agent resolved, all 33 attempted instances",
        "value": 0.7576,
        "unit": "rate",
        "display": "76% (25/33)",
        "n": 33,
        "ci": [
          0.5898,
          0.8717
        ],
        "note": "Both campaigns, one attempt each, failures and empty patches included."
      },
      {
        "id": "agent-rate-c1",
        "label": "Agent resolved, campaign 1 sample of 25",
        "value": 0.72,
        "unit": "rate",
        "display": "72% (18/25)",
        "n": 25,
        "ci": [
          0.5242,
          0.8572
        ]
      },
      {
        "id": "agent-rate-original25",
        "label": "Agent resolved, original seed draw of 25 (no replacements)",
        "value": 0.76,
        "unit": "rate",
        "display": "76% (19/25)",
        "n": 25,
        "ci": [
          0.5657,
          0.885
        ],
        "note": "Mixes two platform builds."
      },
      {
        "id": "panel-mean-33",
        "label": "Public panel mean on the same 33 instances",
        "value": 0.741,
        "unit": "rate",
        "display": "74.1%",
        "n": 33,
        "note": "Mean of 11 public mini-SWE-agent v2 runs."
      },
      {
        "id": "cost-per-attempt",
        "label": "Agent model cost per attempt (notional)",
        "value": 2.81,
        "unit": "usd",
        "display": "$2.81",
        "n": 33,
        "note": "Subscription calls priced at list price; onboarding and failed attempts included."
      },
      {
        "id": "cost-per-resolved",
        "label": "Agent model cost per resolved instance (notional)",
        "value": 3.71,
        "unit": "usd",
        "display": "$3.71",
        "n": 25
      },
      {
        "id": "median-minutes",
        "label": "Median worker time per attempt",
        "value": 9.6,
        "unit": "minutes",
        "display": "9.6 min",
        "n": 33,
        "note": "Range 1.6 to 54.1 min."
      },
      {
        "id": "calls-per-attempt",
        "label": "Model calls per attempt",
        "value": 49.5,
        "unit": "calls",
        "display": "49",
        "n": 33
      }
    ],
    "charts": [
      {
        "id": "swebench-same-instance-leaderboard",
        "title": "Resolved rate on the same 33 SWE-bench Verified instances",
        "subtitle": "Agent vs 11 public mini-SWE-agent v2 runs, one attempt each",
        "kind": "dot-range",
        "unit": "rate",
        "yLabel": "Resolved",
        "series": [
          {
            "name": "Resolved rate",
            "points": [
              {
                "label": "GPT 5.2 (high)",
                "value": 0.8485,
                "lo": 0.6908,
                "hi": 0.9335,
                "n": 33
              },
              {
                "label": "Gemini 3 Flash (high)",
                "value": 0.8182,
                "lo": 0.6561,
                "hi": 0.9139,
                "n": 33
              },
              {
                "label": "GLM 5 (high)",
                "value": 0.7879,
                "lo": 0.6225,
                "hi": 0.8932,
                "n": 33
              },
              {
                "label": "Agent (Sonnet 5.5, full pipeline)",
                "value": 0.7576,
                "lo": 0.5898,
                "hi": 0.8717,
                "n": 33,
                "highlight": true
              },
              {
                "label": "Claude 4.5 Sonnet (high)",
                "value": 0.7576,
                "lo": 0.5898,
                "hi": 0.8717,
                "n": 33
              },
              {
                "label": "Claude 4.5 Haiku (high)",
                "value": 0.7576,
                "lo": 0.5898,
                "hi": 0.8717,
                "n": 33
              },
              {
                "label": "Claude 4.5 Opus (high)",
                "value": 0.7273,
                "lo": 0.5578,
                "hi": 0.8493,
                "n": 33
              },
              {
                "label": "DeepSeek V3.2 (high)",
                "value": 0.7273,
                "lo": 0.5578,
                "hi": 0.8493,
                "n": 33
              },
              {
                "label": "MiniMax M2.5 (high)",
                "value": 0.697,
                "lo": 0.5266,
                "hi": 0.8262,
                "n": 33
              },
              {
                "label": "Claude 4.6 Opus",
                "value": 0.697,
                "lo": 0.5266,
                "hi": 0.8262,
                "n": 33
              },
              {
                "label": "Kimi K2.5 (high)",
                "value": 0.697,
                "lo": 0.5266,
                "hi": 0.8262,
                "n": 33
              },
              {
                "label": "GPT 5 mini",
                "value": 0.6364,
                "lo": 0.4662,
                "hi": 0.7781,
                "n": 33
              }
            ]
          }
        ],
        "note": "Dots show the rate; whiskers show the 95% Wilson interval for n = 33. The panel compares models under one harness; Agent is a full system on one model.",
        "sourceIds": [
          "agent-swebench-c1",
          "agent-swebench-c2",
          "swebench-leaderboard"
        ]
      },
      {
        "id": "swebench-by-difficulty-band",
        "title": "Resolved rate by difficulty band",
        "subtitle": "Band = how many of the 11 public panel models solved the instance",
        "kind": "grouped-bar",
        "unit": "rate",
        "xLabel": "Difficulty band",
        "yLabel": "Resolved",
        "series": [
          {
            "name": "Agent",
            "points": [
              {
                "label": "No panel model solved it",
                "value": 0.25,
                "lo": 0.0456,
                "hi": 0.6994,
                "n": 4,
                "highlight": true
              },
              {
                "label": "Under half solved it",
                "value": 0.75,
                "lo": 0.3006,
                "hi": 0.9544,
                "n": 4,
                "highlight": true
              },
              {
                "label": "Half or more solved it",
                "value": 0.8182,
                "lo": 0.523,
                "hi": 0.9486,
                "n": 11,
                "highlight": true
              },
              {
                "label": "Every panel model solved it",
                "value": 0.8571,
                "lo": 0.6006,
                "hi": 0.9599,
                "n": 14,
                "highlight": true
              }
            ]
          },
          {
            "name": "Public panel mean",
            "points": [
              {
                "label": "No panel model solved it",
                "value": 0,
                "n": 4
              },
              {
                "label": "Under half solved it",
                "value": 0.2727,
                "n": 4
              },
              {
                "label": "Half or more solved it",
                "value": 0.8512,
                "n": 11
              },
              {
                "label": "Every panel model solved it",
                "value": 1,
                "n": 14
              }
            ]
          }
        ],
        "note": "Agent whiskers are 95% Wilson intervals. The \"no panel model solved it\" band has 4 instances; Agent resolved 1 (matplotlib__matplotlib-21568).",
        "sourceIds": [
          "agent-swebench-c1",
          "agent-swebench-c2",
          "swebench-leaderboard",
          "swebench-protocol"
        ]
      },
      {
        "id": "swebench-cost-vs-resolved",
        "title": "Cost per instance vs resolved rate",
        "subtitle": "Same 33 instances. Panel = API list price; Agent = notional subscription estimate",
        "kind": "scatter",
        "unit": "rate",
        "xLabel": "Mean model cost per instance (USD)",
        "yLabel": "Resolved rate",
        "series": [
          {
            "name": "Public panel (mini-SWE-agent v2)",
            "points": [
              {
                "label": "Claude 4.5 Opus (high)",
                "x": 0.861,
                "value": 0.7273,
                "n": 33
              },
              {
                "label": "Gemini 3 Flash (high)",
                "x": 0.357,
                "value": 0.8182,
                "n": 33
              },
              {
                "label": "MiniMax M2.5 (high)",
                "x": 0.075,
                "value": 0.697,
                "n": 33
              },
              {
                "label": "Claude 4.6 Opus",
                "x": 0.61,
                "value": 0.697,
                "n": 33
              },
              {
                "label": "GLM 5 (high)",
                "x": 0.525,
                "value": 0.7879,
                "n": 33
              },
              {
                "label": "GPT 5.2 (high)",
                "x": 0.533,
                "value": 0.8485,
                "n": 33
              },
              {
                "label": "Claude 4.5 Sonnet (high)",
                "x": 0.692,
                "value": 0.7576,
                "n": 33
              },
              {
                "label": "Kimi K2.5 (high)",
                "x": 0.179,
                "value": 0.697,
                "n": 33
              },
              {
                "label": "DeepSeek V3.2 (high)",
                "x": 0.463,
                "value": 0.7273,
                "n": 33
              },
              {
                "label": "Claude 4.5 Haiku (high)",
                "x": 0.363,
                "value": 0.7576,
                "n": 33
              },
              {
                "label": "GPT 5 mini",
                "x": 0.051,
                "value": 0.6364,
                "n": 33
              }
            ]
          },
          {
            "name": "Agent",
            "points": [
              {
                "label": "Agent",
                "x": 2.807,
                "value": 0.7576,
                "n": 33,
                "highlight": true
              }
            ]
          }
        ],
        "note": "Agent's cost includes repository onboarding, planning, verification and review; it is a list-price estimate for subscription calls, not an invoice. Panel costs are published API costs.",
        "sourceIds": [
          "agent-swebench-c1",
          "agent-swebench-c2",
          "swebench-leaderboard",
          "price-anthropic"
        ]
      },
      {
        "id": "swebench-model-calls",
        "title": "Model calls per instance",
        "subtitle": "Mean over the same 33 instances",
        "kind": "bar",
        "unit": "calls",
        "yLabel": "Calls per instance",
        "series": [
          {
            "name": "Mean calls",
            "points": [
              {
                "label": "DeepSeek V3.2 (high)",
                "value": 88.2,
                "n": 33
              },
              {
                "label": "GLM 5 (high)",
                "value": 77.5,
                "n": 33
              },
              {
                "label": "Claude 4.5 Haiku (high)",
                "value": 68.5,
                "n": 33
              },
              {
                "label": "MiniMax M2.5 (high)",
                "value": 58.4,
                "n": 33
              },
              {
                "label": "Kimi K2.5 (high)",
                "value": 56.7,
                "n": 33
              },
              {
                "label": "Gemini 3 Flash (high)",
                "value": 54.2,
                "n": 33
              },
              {
                "label": "Claude 4.5 Sonnet (high)",
                "value": 51,
                "n": 33
              },
              {
                "label": "Agent",
                "value": 49.5,
                "n": 33,
                "highlight": true
              },
              {
                "label": "Claude 4.5 Opus (high)",
                "value": 35.9,
                "n": 33
              },
              {
                "label": "GPT 5.2 (high)",
                "value": 35.6,
                "n": 33
              },
              {
                "label": "Claude 4.6 Opus",
                "value": 28.9,
                "n": 33
              },
              {
                "label": "GPT 5 mini",
                "value": 20.8,
                "n": 33
              }
            ]
          }
        ],
        "note": "A panel call is one bash-agent step. An Agent call is one model request of any stage (research, plan, act, verify, review).",
        "sourceIds": [
          "agent-swebench-c1",
          "agent-swebench-c2",
          "swebench-leaderboard"
        ]
      },
      {
        "id": "swebench-cost-by-stage",
        "title": "Where Agent's model spend goes",
        "subtitle": "Share of notional model cost by stage, all 33 attempts",
        "kind": "bar",
        "unit": "usd",
        "yLabel": "USD (notional)",
        "series": [
          {
            "name": "Cost",
            "points": [
              {
                "label": "Act (edit and run)",
                "value": 44.05
              },
              {
                "label": "Research",
                "value": 20.56
              },
              {
                "label": "Verify",
                "value": 8.72
              },
              {
                "label": "Other",
                "value": 6.24
              },
              {
                "label": "Review",
                "value": 5.58
              },
              {
                "label": "Context compaction",
                "value": 5.41
              },
              {
                "label": "Onboarding notes",
                "value": 2.09
              }
            ]
          }
        ],
        "note": "Total $92.64 over 33 attempts.",
        "sourceIds": [
          "agent-swebench-c1",
          "agent-swebench-c2"
        ]
      },
      {
        "id": "swebench-views",
        "title": "Every way to slice the run, with intervals",
        "subtitle": "Agent resolved rate and 95% Wilson interval per declared view",
        "kind": "dot-range",
        "unit": "rate",
        "yLabel": "Resolved",
        "series": [
          {
            "name": "Agent",
            "points": [
              {
                "label": "Campaign 1: 25-instance sample",
                "value": 0.72,
                "lo": 0.5242,
                "hi": 0.8572,
                "n": 25
              },
              {
                "label": "Campaign 2: 8 compiled-extension instances",
                "value": 0.875,
                "lo": 0.5291,
                "hi": 0.9776,
                "n": 8
              },
              {
                "label": "Original seed draw of 25",
                "value": 0.76,
                "lo": 0.5657,
                "hi": 0.885,
                "n": 25
              },
              {
                "label": "All 33 attempted",
                "value": 0.7576,
                "lo": 0.5898,
                "hi": 0.8717,
                "n": 33,
                "highlight": true
              }
            ]
          },
          {
            "name": "Public panel mean, same instances",
            "points": [
              {
                "label": "Campaign 1: 25-instance sample",
                "value": 0.7091,
                "n": 25
              },
              {
                "label": "Campaign 2: 8 compiled-extension instances",
                "value": 0.8409,
                "n": 8
              },
              {
                "label": "Original seed draw of 25",
                "value": 0.7091,
                "n": 25
              },
              {
                "label": "All 33 attempted",
                "value": 0.741,
                "n": 33
              }
            ]
          }
        ],
        "note": "The original draw and \"all 33\" mix two platform builds.",
        "sourceIds": [
          "agent-swebench-c1",
          "agent-swebench-c2",
          "swebench-leaderboard",
          "swebench-protocol"
        ]
      }
    ],
    "tables": [
      {
        "id": "swebench-per-instance",
        "title": "Every attempt",
        "columns": [
          {
            "key": "instance",
            "label": "Instance",
            "unit": "text"
          },
          {
            "key": "band",
            "label": "Panel solved",
            "unit": "text"
          },
          {
            "key": "outcome",
            "label": "Agent outcome",
            "unit": "text"
          },
          {
            "key": "minutes",
            "label": "Minutes",
            "unit": "minutes"
          },
          {
            "key": "calls",
            "label": "Calls",
            "unit": "calls"
          },
          {
            "key": "costUsd",
            "label": "Cost (notional)",
            "unit": "usd"
          },
          {
            "key": "campaign",
            "label": "Campaign",
            "unit": "text"
          }
        ],
        "rows": [
          {
            "instance": "astropy__astropy-13579",
            "band": "11/11",
            "outcome": "resolved",
            "minutes": 54.1,
            "calls": 65,
            "costUsd": 4,
            "campaign": "2"
          },
          {
            "instance": "astropy__astropy-14096",
            "band": "10/11",
            "outcome": "resolved",
            "minutes": 10.4,
            "calls": 64,
            "costUsd": 3.35,
            "campaign": "2"
          },
          {
            "instance": "astropy__astropy-7336",
            "band": "11/11",
            "outcome": "unresolved",
            "minutes": 23,
            "calls": 48,
            "costUsd": 2.03,
            "campaign": "2"
          },
          {
            "instance": "django__django-10554",
            "band": "0/11",
            "outcome": "unresolved",
            "minutes": 4.7,
            "calls": 34,
            "costUsd": 1.52,
            "campaign": "1"
          },
          {
            "instance": "django__django-11333",
            "band": "11/11",
            "outcome": "resolved",
            "minutes": 8,
            "calls": 46,
            "costUsd": 2.51,
            "campaign": "1"
          },
          {
            "instance": "django__django-11885",
            "band": "3/11",
            "outcome": "resolved",
            "minutes": 7.3,
            "calls": 38,
            "costUsd": 2.27,
            "campaign": "1"
          },
          {
            "instance": "django__django-12193",
            "band": "10/11",
            "outcome": "resolved",
            "minutes": 5.7,
            "calls": 43,
            "costUsd": 2.06,
            "campaign": "1"
          },
          {
            "instance": "django__django-12741",
            "band": "11/11",
            "outcome": "resolved",
            "minutes": 10.5,
            "calls": 54,
            "costUsd": 2.98,
            "campaign": "1"
          },
          {
            "instance": "django__django-13158",
            "band": "10/11",
            "outcome": "resolved",
            "minutes": 7,
            "calls": 43,
            "costUsd": 2.15,
            "campaign": "1"
          },
          {
            "instance": "django__django-13925",
            "band": "8/11",
            "outcome": "resolved",
            "minutes": 6.8,
            "calls": 43,
            "costUsd": 2.29,
            "campaign": "1"
          },
          {
            "instance": "django__django-14034",
            "band": "0/11",
            "outcome": "unresolved",
            "minutes": 11.5,
            "calls": 54,
            "costUsd": 3.03,
            "campaign": "1"
          },
          {
            "instance": "django__django-14373",
            "band": "11/11",
            "outcome": "resolved",
            "minutes": 5.7,
            "calls": 38,
            "costUsd": 1.74,
            "campaign": "1"
          },
          {
            "instance": "django__django-14559",
            "band": "11/11",
            "outcome": "resolved",
            "minutes": 7.2,
            "calls": 47,
            "costUsd": 2.71,
            "campaign": "1"
          },
          {
            "instance": "django__django-14631",
            "band": "9/11",
            "outcome": "empty patch",
            "minutes": 38.2,
            "calls": 47,
            "costUsd": 2.8,
            "campaign": "1"
          },
          {
            "instance": "django__django-14672",
            "band": "11/11",
            "outcome": "resolved",
            "minutes": 9.1,
            "calls": 58,
            "costUsd": 2.99,
            "campaign": "1"
          },
          {
            "instance": "django__django-15022",
            "band": "4/11",
            "outcome": "unresolved",
            "minutes": 15,
            "calls": 62,
            "costUsd": 4.1,
            "campaign": "1"
          },
          {
            "instance": "django__django-15280",
            "band": "7/11",
            "outcome": "resolved",
            "minutes": 11,
            "calls": 59,
            "costUsd": 3.67,
            "campaign": "1"
          },
          {
            "instance": "django__django-15572",
            "band": "10/11",
            "outcome": "resolved",
            "minutes": 5.3,
            "calls": 42,
            "costUsd": 1.81,
            "campaign": "1"
          },
          {
            "instance": "django__django-15732",
            "band": "3/11",
            "outcome": "resolved",
            "minutes": 10.9,
            "calls": 59,
            "costUsd": 2.9,
            "campaign": "1"
          },
          {
            "instance": "django__django-16255",
            "band": "10/11",
            "outcome": "resolved",
            "minutes": 7.6,
            "calls": 47,
            "costUsd": 2.55,
            "campaign": "1"
          },
          {
            "instance": "django__django-16333",
            "band": "11/11",
            "outcome": "resolved",
            "minutes": 13,
            "calls": 37,
            "costUsd": 1.61,
            "campaign": "1"
          },
          {
            "instance": "django__django-17029",
            "band": "11/11",
            "outcome": "resolved",
            "minutes": 7.8,
            "calls": 42,
            "costUsd": 2.28,
            "campaign": "1"
          },
          {
            "instance": "matplotlib__matplotlib-21568",
            "band": "0/11",
            "outcome": "resolved",
            "minutes": 9.4,
            "calls": 53,
            "costUsd": 3.02,
            "campaign": "2"
          },
          {
            "instance": "matplotlib__matplotlib-24149",
            "band": "10/11",
            "outcome": "resolved",
            "minutes": 41.1,
            "calls": 64,
            "costUsd": 3.64,
            "campaign": "2"
          },
          {
            "instance": "matplotlib__matplotlib-24970",
            "band": "11/11",
            "outcome": "resolved",
            "minutes": 33,
            "calls": 66,
            "costUsd": 3.9,
            "campaign": "2"
          },
          {
            "instance": "pydata__xarray-6721",
            "band": "11/11",
            "outcome": "resolved",
            "minutes": 16.2,
            "calls": 57,
            "costUsd": 4.09,
            "campaign": "1"
          },
          {
            "instance": "scikit-learn__scikit-learn-14894",
            "band": "11/11",
            "outcome": "resolved",
            "minutes": 35.2,
            "calls": 50,
            "costUsd": 3.23,
            "campaign": "2"
          },
          {
            "instance": "scikit-learn__scikit-learn-25973",
            "band": "10/11",
            "outcome": "resolved",
            "minutes": 8.6,
            "calls": 44,
            "costUsd": 2.43,
            "campaign": "2"
          },
          {
            "instance": "sphinx-doc__sphinx-10435",
            "band": "2/11",
            "outcome": "resolved",
            "minutes": 16.8,
            "calls": 67,
            "costUsd": 4.83,
            "campaign": "1"
          },
          {
            "instance": "sphinx-doc__sphinx-9698",
            "band": "11/11",
            "outcome": "resolved",
            "minutes": 9.6,
            "calls": 49,
            "costUsd": 2.69,
            "campaign": "1"
          },
          {
            "instance": "sympy__sympy-18189",
            "band": "11/11",
            "outcome": "empty patch",
            "minutes": 1.6,
            "calls": 13,
            "costUsd": 0.89,
            "campaign": "1"
          },
          {
            "instance": "sympy__sympy-20428",
            "band": "0/11",
            "outcome": "unresolved",
            "minutes": 21,
            "calls": 71,
            "costUsd": 4.81,
            "campaign": "1"
          },
          {
            "instance": "sympy__sympy-22456",
            "band": "9/11",
            "outcome": "empty patch",
            "minutes": 6.4,
            "calls": 28,
            "costUsd": 1.75,
            "campaign": "1"
          }
        ]
      },
      {
        "id": "swebench-panel",
        "title": "Public panel on the same instances",
        "columns": [
          {
            "key": "system",
            "label": "System",
            "unit": "text"
          },
          {
            "key": "resolved",
            "label": "Resolved of 33",
            "unit": "count"
          },
          {
            "key": "rate",
            "label": "Rate",
            "unit": "rate"
          },
          {
            "key": "board500",
            "label": "Full Verified board (500)",
            "unit": "percent"
          },
          {
            "key": "meanCost",
            "label": "Mean cost per instance, these 33",
            "unit": "usd"
          }
        ],
        "rows": [
          {
            "system": "GPT 5.2 (high)",
            "resolved": 28,
            "rate": 0.8485,
            "board500": 72.8,
            "meanCost": 0.533
          },
          {
            "system": "Gemini 3 Flash (high)",
            "resolved": 27,
            "rate": 0.8182,
            "board500": 75.8,
            "meanCost": 0.357
          },
          {
            "system": "GLM 5 (high)",
            "resolved": 26,
            "rate": 0.7879,
            "board500": 72.8,
            "meanCost": 0.525
          },
          {
            "system": "Agent (Sonnet 5.5, full pipeline)",
            "resolved": 25,
            "rate": 0.7576,
            "board500": null,
            "meanCost": 2.81
          },
          {
            "system": "Claude 4.5 Sonnet (high)",
            "resolved": 25,
            "rate": 0.7576,
            "board500": 71.4,
            "meanCost": 0.692
          },
          {
            "system": "Claude 4.5 Haiku (high)",
            "resolved": 25,
            "rate": 0.7576,
            "board500": 66.6,
            "meanCost": 0.363
          },
          {
            "system": "Claude 4.5 Opus (high)",
            "resolved": 24,
            "rate": 0.7273,
            "board500": 76.8,
            "meanCost": 0.861
          },
          {
            "system": "DeepSeek V3.2 (high)",
            "resolved": 24,
            "rate": 0.7273,
            "board500": 70,
            "meanCost": 0.463
          },
          {
            "system": "MiniMax M2.5 (high)",
            "resolved": 23,
            "rate": 0.697,
            "board500": 75.8,
            "meanCost": 0.075
          },
          {
            "system": "Claude 4.6 Opus",
            "resolved": 23,
            "rate": 0.697,
            "board500": 75.6,
            "meanCost": 0.61
          },
          {
            "system": "Kimi K2.5 (high)",
            "resolved": 23,
            "rate": 0.697,
            "board500": 70.8,
            "meanCost": 0.179
          },
          {
            "system": "GPT 5 mini",
            "resolved": 21,
            "rate": 0.6364,
            "board500": 56.2,
            "meanCost": 0.051
          }
        ]
      }
    ]
  },
  "sources": [
    {
      "id": "agent-swebench-c1",
      "title": "Agent on SWE-bench Verified, campaign 1 (25 instances)",
      "kind": "run",
      "date": "2026-10-04",
      "note": "Stratified sample of 25 Verified instances (seed 20261004), one attempt each, official grading harness. Fixed model claude-sonnet-5-5, platform build f0ac3a8a.",
      "data": [
        "/benchmarks/raw/swebench/attempts.json",
        "/benchmarks/raw/swebench/exclusions.json"
      ]
    },
    {
      "id": "agent-swebench-c2",
      "title": "Agent on SWE-bench Verified, campaign 2 (8 compiled-extension instances)",
      "kind": "run",
      "date": "2026-10-05",
      "note": "The 6 compiled-extension instances that campaign 1 could not run, plus 2 replacement candidates. One attempt each, platform build 236c0d3f.",
      "data": [
        "/benchmarks/raw/swebench/attempts.json"
      ]
    },
    {
      "id": "swebench-leaderboard",
      "title": "SWE-bench Verified leaderboard, mini-SWE-agent v2 runs",
      "kind": "public-leaderboard",
      "date": "2026-02-17",
      "url": "https://www.swebench.com",
      "note": "Public per-instance results of 11 models under mini-SWE-agent 2.0.0 (bash only, one attempt). Costs are API list prices as published.",
      "data": [
        "/benchmarks/raw/swebench/panel.json"
      ]
    },
    {
      "id": "swebench-protocol",
      "title": "SWE-bench campaign rules and sample design",
      "kind": "protocol",
      "date": "2026-10-04",
      "note": "Rules declared before the first run: escalations are graded as delivered, gold must resolve on the host, blocked instances are replaced in the same difficulty band, no second attempts. The excluded and replaced instances are listed in the exclusions extract.",
      "data": [
        "/benchmarks/raw/swebench/exclusions.json"
      ]
    },
    {
      "id": "price-anthropic",
      "title": "Anthropic list prices (Claude models)",
      "kind": "price-list",
      "date": "2026-09-21",
      "url": "https://platform.claude.com/docs/en/about-claude/pricing",
      "note": "Prices as listed by the vendor on 2026-09-21 and recorded in the product price table. Cache reads at the listed rate, one-hour cache writes at twice the input price."
    }
  ]
}
