{
  "schema": "agent-public-bench@1",
  "generatedAt": "2026-10-07T00:00:00.000Z",
  "url": "https://agent.sasid.ai/benchmarks/cost-thought-experiments",
  "study": {
    "slug": "cost-thought-experiments",
    "title": "What if every call ran on Opus? Repricing real agent tokens",
    "seoTitle": "Agent token costs repriced: Haiku vs Sonnet vs Opus vs Fable",
    "description": "Thought experiments on real tokens: the same SWE-bench agent work priced at Haiku, Sonnet, Opus, Fable, Gemini Flash, GPT and Jev list prices.",
    "question": "Agent recorded every token it used on 33 SWE-bench instances. What would the same tokens cost at other models’ list prices, and what did caching save?",
    "answer": "Calculation, not a run. Agent used 162.9M input tokens (94.0% cache reads) and 1.8M output tokens across 33 attempts. At Sonnet 5.5 list prices that is $87.23 ($3.49 per resolved instance); the platform's own notional figure, which also counts compaction calls, is $92.64. The same tokens at Opus 5.5 prices cost $143.83, at Fable 5.1 $321.31 and at Haiku 4.5 $43.61. Without prompt caching the Sonnet bill would be $343.33. A different model would have used different tokens and resolved a different set, so these figures bound price sensitivity; they do not predict outcomes.",
    "date": "2026-10-05",
    "updated": "2026-10-05",
    "tags": [
      "thought-experiment",
      "llm-pricing",
      "prompt-caching",
      "opus",
      "sonnet",
      "haiku",
      "jev"
    ],
    "method": [
      "Tokens: the sum over every model call in the run telemetry of all 33 SWE-bench attempts (input, cache reads, one-hour cache writes, output).",
      "Prices: list prices recorded in the product price table, effective 2026-09-21 (Jev 2026-09-23, OpenAI rows 2026-10-03).",
      "Formula: uncached input × input price + cache reads × cache-read price + cache writes × write price + output × output price. Anthropic one-hour writes cost twice the input price; other vendors’ writes are priced as plain input.",
      "Cost per resolved keeps every attempt’s cost in the numerator and divides by the 25 instances Agent resolved."
    ],
    "caveats": [
      "Every repriced figure is a calculation, not a run. Only the Sonnet 5.5 row matches the model that produced the tokens.",
      "Different models use different numbers of calls, tokens and cache hits, and they resolve different instances. Use these figures for price sensitivity only.",
      "Jev is a routing model. Pricing coding tokens at Jev rates shows a floor, not a feasible configuration.",
      "The router-overhead chart uses assumed decision prompt sizes.",
      "Recorded costs are list-price estimates for subscription calls; no invoice backs them."
    ],
    "sourceIds": [
      "calc-repricing",
      "agent-swebench-c1",
      "agent-swebench-c2",
      "swebench-leaderboard",
      "price-anthropic",
      "price-google",
      "price-openai",
      "price-jev"
    ],
    "stats": [
      {
        "id": "tokens-input",
        "label": "Input tokens recorded",
        "value": 162861253,
        "unit": "tokens",
        "display": "162.9M",
        "n": 33
      },
      {
        "id": "tokens-output",
        "label": "Output tokens recorded",
        "value": 1760652,
        "unit": "tokens",
        "display": "1.8M",
        "n": 33
      },
      {
        "id": "cache-read-share",
        "label": "Share of input served from cache",
        "value": 0.9401,
        "unit": "rate",
        "display": "94.0%",
        "n": 33
      },
      {
        "id": "sonnet-repriced",
        "label": "Recorded tokens at Sonnet 5.5 list price",
        "value": 87.23,
        "unit": "usd",
        "display": "$87.23",
        "n": 33
      },
      {
        "id": "opus-repriced",
        "label": "Same tokens at Opus 5.5 list price (calculation)",
        "value": 143.83,
        "unit": "usd",
        "display": "$143.83",
        "n": 33
      },
      {
        "id": "haiku-repriced",
        "label": "Same tokens at Haiku 4.5 list price (calculation)",
        "value": 43.61,
        "unit": "usd",
        "display": "$43.61",
        "n": 33
      },
      {
        "id": "no-cache-sonnet",
        "label": "Sonnet 5.5 without caching (calculation)",
        "value": 343.33,
        "unit": "usd",
        "display": "$343.33",
        "n": 33
      },
      {
        "id": "panel-cost-per-resolved-mean",
        "label": "Public panel mean cost per resolved instance (recorded)",
        "value": 0.569,
        "unit": "usd",
        "display": "$0.57",
        "n": 11
      }
    ],
    "charts": [
      {
        "id": "repriced-cost-per-resolved",
        "title": "Thought experiment: the same tokens at other list prices",
        "subtitle": "Cost per resolved SWE-bench instance if 162.9M input and 1.8M output tokens had been billed at each model's list price",
        "kind": "bar",
        "unit": "usd",
        "yLabel": "USD per resolved instance",
        "series": [
          {
            "name": "Repriced cost per resolved instance",
            "points": [
              {
                "label": "Claude Fable 5.1",
                "value": 12.852,
                "highlight": false
              },
              {
                "label": "Claude Opus 5",
                "value": 8.723,
                "highlight": false
              },
              {
                "label": "Claude Opus 5.5",
                "value": 5.753,
                "highlight": false
              },
              {
                "label": "Claude Sonnet 5.5",
                "value": 3.489,
                "highlight": true
              },
              {
                "label": "GPT-6.1 Sol",
                "value": 2.097,
                "highlight": false
              },
              {
                "label": "Claude Haiku 4.5",
                "value": 1.745,
                "highlight": false
              },
              {
                "label": "Gemini 3.x Flash",
                "value": 1.016,
                "highlight": false
              },
              {
                "label": "Jev 1.13 (router)",
                "value": 0.042,
                "highlight": false
              }
            ]
          }
        ],
        "note": "Calculation, not a run: tokens recorded by Agent on claude-sonnet-5-5 (33 attempts, 25 resolved) times list prices effective 2026-09-21. Another model would use a different number of tokens and resolve a different set. Jev is a routing model and cannot do this work; its bar is a price floor only.",
        "sourceIds": [
          "calc-repricing",
          "agent-swebench-c1",
          "agent-swebench-c2",
          "price-anthropic",
          "price-google",
          "price-openai",
          "price-jev"
        ]
      },
      {
        "id": "cost-per-resolved-agent-vs-panel",
        "title": "Recorded cost per resolved instance: Agent vs the public panel",
        "subtitle": "Same 33 SWE-bench Verified instances; all attempts in the numerator",
        "kind": "bar",
        "unit": "usd",
        "yLabel": "USD per resolved instance",
        "series": [
          {
            "name": "Cost per resolved instance",
            "points": [
              {
                "label": "Agent (notional)",
                "value": 3.706,
                "n": 25,
                "highlight": true
              },
              {
                "label": "Claude 4.5 Opus (high)",
                "value": 1.184,
                "n": 24
              },
              {
                "label": "Claude 4.5 Sonnet (high)",
                "value": 0.913,
                "n": 25
              },
              {
                "label": "Claude 4.6 Opus",
                "value": 0.875,
                "n": 23
              },
              {
                "label": "GLM 5 (high)",
                "value": 0.667,
                "n": 26
              },
              {
                "label": "DeepSeek V3.2 (high)",
                "value": 0.637,
                "n": 24
              },
              {
                "label": "GPT 5.2 (high)",
                "value": 0.628,
                "n": 28
              },
              {
                "label": "Claude 4.5 Haiku (high)",
                "value": 0.479,
                "n": 25
              },
              {
                "label": "Gemini 3 Flash (high)",
                "value": 0.436,
                "n": 27
              },
              {
                "label": "Kimi K2.5 (high)",
                "value": 0.256,
                "n": 23
              },
              {
                "label": "MiniMax M2.5 (high)",
                "value": 0.107,
                "n": 23
              },
              {
                "label": "GPT 5 mini",
                "value": 0.08,
                "n": 21
              }
            ]
          }
        ],
        "note": "Recorded figures, not repricing. Panel costs are published API costs for a bash-only agent. Agent's figure is a list-price estimate of subscription calls and includes onboarding, planning, verification and review.",
        "sourceIds": [
          "agent-swebench-c1",
          "agent-swebench-c2",
          "swebench-leaderboard"
        ]
      },
      {
        "id": "prompt-cache-savings",
        "title": "Thought experiment: what prompt caching saved",
        "subtitle": "The same recorded tokens with and without cache pricing",
        "kind": "grouped-bar",
        "unit": "usd",
        "yLabel": "USD for all attempts",
        "series": [
          {
            "name": "With caching (as recorded)",
            "points": [
              {
                "label": "Claude Haiku 4.5",
                "value": 43.61,
                "highlight": false
              },
              {
                "label": "Claude Sonnet 5.5",
                "value": 87.23,
                "highlight": true
              },
              {
                "label": "Claude Opus 5.5",
                "value": 143.83,
                "highlight": false
              },
              {
                "label": "Claude Fable 5.1",
                "value": 321.31,
                "highlight": false
              }
            ]
          },
          {
            "name": "Without caching",
            "points": [
              {
                "label": "Claude Haiku 4.5",
                "value": 171.66
              },
              {
                "label": "Claude Sonnet 5.5",
                "value": 343.33
              },
              {
                "label": "Claude Opus 5.5",
                "value": 686.66
              },
              {
                "label": "Claude Fable 5.1",
                "value": 1716.65
              }
            ]
          }
        ],
        "note": "94.0% of recorded input tokens were cache reads. Calculation, not a run.",
        "sourceIds": [
          "calc-repricing",
          "agent-swebench-c1",
          "agent-swebench-c2",
          "price-anthropic"
        ]
      },
      {
        "id": "token-cost-mix",
        "title": "Where the token dollars go",
        "subtitle": "Recorded tokens at Sonnet 5.5 list price, by token kind",
        "kind": "bar",
        "unit": "usd",
        "yLabel": "USD for all attempts",
        "series": [
          {
            "name": "Cost",
            "points": [
              {
                "label": "Cache writes (1 h)",
                "value": 38.99
              },
              {
                "label": "Cache reads",
                "value": 30.62
              },
              {
                "label": "Output",
                "value": 17.61
              },
              {
                "label": "Uncached input",
                "value": 0.01
              }
            ]
          }
        ],
        "note": "List-price calculation on recorded tokens: 153.1M cache reads, 9.7M cache writes, 1.8M output, 3.2k uncached input. An agent loop re-reads its context on every call, so cache reads dominate the token count; by price the largest part is cache writes (1 h).",
        "sourceIds": [
          "calc-repricing",
          "agent-swebench-c1",
          "agent-swebench-c2",
          "price-anthropic"
        ]
      },
      {
        "id": "router-overhead-jev",
        "title": "Thought experiment: the price of a routing decision on every call",
        "subtitle": "1,632 model calls; one Jev decision per call at an assumed prompt size",
        "kind": "bar",
        "unit": "usd",
        "yLabel": "USD for all attempts",
        "series": [
          {
            "name": "Jev decision cost",
            "points": [
              {
                "label": "1k-token decision prompt",
                "value": 0.0685
              },
              {
                "label": "2k-token decision prompt",
                "value": 0.1371
              },
              {
                "label": "5k-token decision prompt",
                "value": 0.3427
              }
            ]
          }
        ],
        "note": "Assumption-based calculation: the decision prompt sizes are assumptions, not measurements. For scale, the recorded work cost $92.64 (notional). A router only pays off if its choices save more than this.",
        "sourceIds": [
          "calc-repricing",
          "price-jev",
          "agent-swebench-c1",
          "agent-swebench-c2"
        ]
      }
    ],
    "tables": [
      {
        "id": "repricing-table",
        "title": "Repricing table (calculation)",
        "columns": [
          {
            "key": "model",
            "label": "Priced as",
            "unit": "text"
          },
          {
            "key": "vendor",
            "label": "Vendor",
            "unit": "text"
          },
          {
            "key": "inputPerM",
            "label": "Input $/M",
            "unit": "usd"
          },
          {
            "key": "cacheReadPerM",
            "label": "Cache read $/M",
            "unit": "usd"
          },
          {
            "key": "outputPerM",
            "label": "Output $/M",
            "unit": "usd"
          },
          {
            "key": "total",
            "label": "All 33 attempts",
            "unit": "usd"
          },
          {
            "key": "perAttempt",
            "label": "Per attempt",
            "unit": "usd"
          },
          {
            "key": "perResolved",
            "label": "Per resolved",
            "unit": "usd"
          }
        ],
        "rows": [
          {
            "model": "Claude Haiku 4.5",
            "vendor": "Anthropic",
            "inputPerM": 1,
            "cacheReadPerM": 0.1,
            "outputPerM": 5,
            "total": 43.61,
            "perAttempt": 1.322,
            "perResolved": 1.745
          },
          {
            "model": "Claude Sonnet 5.5",
            "vendor": "Anthropic",
            "inputPerM": 2,
            "cacheReadPerM": 0.2,
            "outputPerM": 10,
            "total": 87.23,
            "perAttempt": 2.643,
            "perResolved": 3.489
          },
          {
            "model": "Claude Opus 5.5",
            "vendor": "Anthropic",
            "inputPerM": 4,
            "cacheReadPerM": 0.2,
            "outputPerM": 20,
            "total": 143.83,
            "perAttempt": 4.359,
            "perResolved": 5.753
          },
          {
            "model": "Claude Opus 5",
            "vendor": "Anthropic",
            "inputPerM": 5,
            "cacheReadPerM": 0.5,
            "outputPerM": 25,
            "total": 218.07,
            "perAttempt": 6.608,
            "perResolved": 8.723
          },
          {
            "model": "Claude Fable 5.1",
            "vendor": "Anthropic",
            "inputPerM": 10,
            "cacheReadPerM": 0.25,
            "outputPerM": 50,
            "total": 321.31,
            "perAttempt": 9.737,
            "perResolved": 12.852
          },
          {
            "model": "Gemini 3.x Flash",
            "vendor": "Google",
            "inputPerM": 0.75,
            "cacheReadPerM": 0.075,
            "outputPerM": 3.75,
            "total": 25.4,
            "perAttempt": 0.77,
            "perResolved": 1.016
          },
          {
            "model": "GPT-6.1 Sol",
            "vendor": "OpenAI",
            "inputPerM": 2,
            "cacheReadPerM": 0.1,
            "outputPerM": 10,
            "total": 52.42,
            "perAttempt": 1.589,
            "perResolved": 2.097
          },
          {
            "model": "Jev 1.13 (router)",
            "vendor": "TypeSafe",
            "inputPerM": 0.042,
            "cacheReadPerM": 0.0042,
            "outputPerM": 0,
            "total": 1.05,
            "perAttempt": 0.032,
            "perResolved": 0.042
          }
        ]
      }
    ]
  },
  "sources": [
    {
      "id": "agent-swebench-c1",
      "title": "Agent on SWE-bench Verified, campaign 1 (25 instances)",
      "kind": "run",
      "date": "2026-10-04",
      "note": "Stratified sample of 25 Verified instances (seed 20261004), one attempt each, official grading harness. Fixed model claude-sonnet-5-5, platform build f0ac3a8a.",
      "data": [
        "/benchmarks/raw/swebench/attempts.json",
        "/benchmarks/raw/swebench/exclusions.json"
      ]
    },
    {
      "id": "agent-swebench-c2",
      "title": "Agent on SWE-bench Verified, campaign 2 (8 compiled-extension instances)",
      "kind": "run",
      "date": "2026-10-05",
      "note": "The 6 compiled-extension instances that campaign 1 could not run, plus 2 replacement candidates. One attempt each, platform build 236c0d3f.",
      "data": [
        "/benchmarks/raw/swebench/attempts.json"
      ]
    },
    {
      "id": "swebench-leaderboard",
      "title": "SWE-bench Verified leaderboard, mini-SWE-agent v2 runs",
      "kind": "public-leaderboard",
      "date": "2026-02-17",
      "url": "https://www.swebench.com",
      "note": "Public per-instance results of 11 models under mini-SWE-agent 2.0.0 (bash only, one attempt). Costs are API list prices as published.",
      "data": [
        "/benchmarks/raw/swebench/panel.json"
      ]
    },
    {
      "id": "price-anthropic",
      "title": "Anthropic list prices (Claude models)",
      "kind": "price-list",
      "date": "2026-09-21",
      "url": "https://platform.claude.com/docs/en/about-claude/pricing",
      "note": "Prices as listed by the vendor on 2026-09-21 and recorded in the product price table. Cache reads at the listed rate, one-hour cache writes at twice the input price."
    },
    {
      "id": "price-google",
      "title": "Google Gemini list prices",
      "kind": "price-list",
      "date": "2026-09-21",
      "url": "https://ai.google.dev/pricing",
      "note": "Gemini 3.x Flash prices as listed by the vendor on 2026-09-21. The vendor announced a doubling from 2027-01-01."
    },
    {
      "id": "price-openai",
      "title": "OpenAI list prices",
      "kind": "price-list",
      "date": "2026-10-03",
      "url": "https://developers.openai.com/api/docs/pricing",
      "note": "Token prices as listed by the vendor on 2026-10-03."
    },
    {
      "id": "price-jev",
      "title": "Jev 1.13 list price",
      "kind": "price-list",
      "date": "2026-09-23",
      "url": "https://docs.typesafe.ai/models",
      "note": "Prices as listed by the vendor on 2026-09-23: input tokens only, output tokens free."
    },
    {
      "id": "calc-repricing",
      "title": "Repricing calculation",
      "kind": "calculation",
      "date": "2026-10-05",
      "note": "Recorded token counts multiplied by the list prices in the price-list sources above. A calculation, not a run: a different model would have used a different number of tokens and reached different outcomes."
    }
  ]
}
