{
  "schema": "agent-public-bench@1",
  "generatedAt": "2026-10-07T00:00:00.000Z",
  "url": "https://agent.sasid.ai/benchmarks/coding-calibration",
  "study": {
    "slug": "coding-calibration",
    "title": "Coding calibration: what broke on three real pull requests",
    "seoTitle": "AI coding agent calibration on fastify, h3 and uvicorn tasks",
    "description": "One attempt per task, four platform builds, failures kept: how an AI worker did on real fastify/session, h3 and uvicorn issues, and what broke.",
    "question": "On three real upstream issues, does the AI worker deliver a verified change, and what stops it when it does not?",
    "answer": "Across four platform builds, verified delivery went from 0 of 3 to 1 of 3; the latest build delivered fastify/session cleanly, but h3 regressed to an empty patch when a formatting failure was lost across a context fold, and uvicorn passed its full suite yet stayed unverified for two gate-classification reasons. The latest slice cost $11.06 for the three tasks (notional). The first calibration run on 2026-10-02 produced a passing patch for fastify/session but stopped at its $5 cap before delivery. One attempt per cell finds defects; it does not measure a rate.",
    "date": "2026-10-02",
    "updated": "2026-10-05",
    "tags": [
      "calibration",
      "coding-agents",
      "failures",
      "fastify",
      "h3",
      "uvicorn"
    ],
    "method": [
      "Tasks: fastify/session #348, h3js/h3 #1533 and Kludex/uvicorn #3036, each frozen at the base commit with the merged change as reference.",
      "Fixed model claude-sonnet-5-5, routing off, balanced mode, real cold onboarding, no review replay, no operator answers.",
      "Worker and gates run offline with frozen dependencies. Gates run install, lint, typecheck, build and the full suite at base, candidate and reference, plus cross-tests.",
      "\"Functional pass\" means the frozen patch passed the gates. \"Verified delivery\" also needs a completed, clean, unassisted run.",
      "One attempt per task per slice. Every stop and failure is kept; nothing is rerun or replaced."
    ],
    "caveats": [
      "One attempt per cell: these are defect-finding runs, not rates.",
      "Each slice changes the platform build, and the last two also remove the caps. No two slices are a matched comparison.",
      "Costs are list-price estimates for subscription calls, not invoices.",
      "The f0ac3a8a h3 run was verified only after its account-route label was corrected; it counts as unverified here, as declared."
    ],
    "sourceIds": [
      "agent-coding-calibration"
    ],
    "stats": [
      {
        "id": "verified-latest",
        "label": "Verified deliveries, latest build",
        "value": 1,
        "unit": "count",
        "display": "1 of 3",
        "n": 3
      },
      {
        "id": "cost-latest",
        "label": "Notional cost, latest build, all 3 tasks",
        "value": 11.06,
        "unit": "usd",
        "display": "$11.06",
        "n": 3
      },
      {
        "id": "refusals-trend",
        "label": "Guardrail refusals, first vs latest slice",
        "value": 19,
        "unit": "count",
        "display": "26 → 19",
        "n": 3
      },
      {
        "id": "first-run-cost",
        "label": "First calibration run (capped, fastify/session)",
        "value": 4.89,
        "unit": "usd",
        "display": "$4.89, stopped at cap",
        "n": 1
      }
    ],
    "charts": [
      {
        "id": "calibration-outcomes-by-slice",
        "title": "Three real tasks, four platform builds",
        "subtitle": "Tasks per slice: functional pass, verified delivery, pull request opened",
        "kind": "grouped-bar",
        "unit": "count",
        "yLabel": "Tasks (of 3)",
        "series": [
          {
            "name": "Functional pass (offline gates)",
            "points": [
              {
                "label": "Baseline (capped)",
                "value": 2,
                "n": 3
              },
              {
                "label": "Fix wave 1 (capped)",
                "value": 2,
                "n": 3
              },
              {
                "label": "Uncapped, build f0ac3a8a",
                "value": 1,
                "n": 3
              },
              {
                "label": "Uncapped, build 236c0d3f",
                "value": 1,
                "n": 3
              }
            ]
          },
          {
            "name": "Verified delivery",
            "points": [
              {
                "label": "Baseline (capped)",
                "value": 0,
                "n": 3,
                "highlight": false
              },
              {
                "label": "Fix wave 1 (capped)",
                "value": 0,
                "n": 3,
                "highlight": false
              },
              {
                "label": "Uncapped, build f0ac3a8a",
                "value": 0,
                "n": 3,
                "highlight": false
              },
              {
                "label": "Uncapped, build 236c0d3f",
                "value": 1,
                "n": 3,
                "highlight": true
              }
            ]
          },
          {
            "name": "Pull request opened",
            "points": [
              {
                "label": "Baseline (capped)",
                "value": 1,
                "n": 3
              },
              {
                "label": "Fix wave 1 (capped)",
                "value": 2,
                "n": 3
              },
              {
                "label": "Uncapped, build f0ac3a8a",
                "value": 2,
                "n": 3
              },
              {
                "label": "Uncapped, build 236c0d3f",
                "value": 2,
                "n": 3
              }
            ]
          }
        ],
        "note": "One attempt per task per slice. The two capped slices stopped at 20 minutes or $5; the uncapped slices had no ceiling. Each slice is a different platform build, so a change is not a matched improvement.",
        "sourceIds": [
          "agent-coding-calibration"
        ]
      },
      {
        "id": "calibration-cost-by-task",
        "title": "Notional model cost per task, by slice",
        "kind": "grouped-bar",
        "unit": "usd",
        "yLabel": "USD (notional)",
        "series": [
          {
            "name": "Baseline (capped)",
            "points": [
              {
                "label": "fastify/session",
                "value": 3.43,
                "highlight": false
              },
              {
                "label": "h3js/h3",
                "value": 4,
                "highlight": false
              },
              {
                "label": "Kludex/uvicorn",
                "value": 4.88,
                "highlight": false
              }
            ]
          },
          {
            "name": "Fix wave 1 (capped)",
            "points": [
              {
                "label": "fastify/session",
                "value": 4.85,
                "highlight": false
              },
              {
                "label": "h3js/h3",
                "value": 3.29,
                "highlight": false
              },
              {
                "label": "Kludex/uvicorn",
                "value": 4.63,
                "highlight": false
              }
            ]
          },
          {
            "name": "Uncapped, build f0ac3a8a",
            "points": [
              {
                "label": "fastify/session",
                "value": 2.44,
                "highlight": false
              },
              {
                "label": "h3js/h3",
                "value": 3.81,
                "highlight": false
              },
              {
                "label": "Kludex/uvicorn",
                "value": 4.74,
                "highlight": false
              }
            ]
          },
          {
            "name": "Uncapped, build 236c0d3f",
            "points": [
              {
                "label": "fastify/session",
                "value": 3.53,
                "highlight": true
              },
              {
                "label": "h3js/h3",
                "value": 2.96,
                "highlight": true
              },
              {
                "label": "Kludex/uvicorn",
                "value": 4.57,
                "highlight": true
              }
            ]
          }
        ],
        "note": "List-price estimates of subscription calls. Failed and capped attempts count.",
        "sourceIds": [
          "agent-coding-calibration"
        ]
      },
      {
        "id": "calibration-minutes-by-task",
        "title": "Wall time per task, by slice",
        "kind": "grouped-bar",
        "unit": "minutes",
        "yLabel": "Minutes",
        "series": [
          {
            "name": "Baseline (capped)",
            "points": [
              {
                "label": "fastify/session",
                "value": 9,
                "highlight": false
              },
              {
                "label": "h3js/h3",
                "value": 11.8,
                "highlight": false
              },
              {
                "label": "Kludex/uvicorn",
                "value": 17.9,
                "highlight": false
              }
            ]
          },
          {
            "name": "Fix wave 1 (capped)",
            "points": [
              {
                "label": "fastify/session",
                "value": 15.5,
                "highlight": false
              },
              {
                "label": "h3js/h3",
                "value": 13.5,
                "highlight": false
              },
              {
                "label": "Kludex/uvicorn",
                "value": 20.3,
                "highlight": false
              }
            ]
          },
          {
            "name": "Uncapped, build f0ac3a8a",
            "points": [
              {
                "label": "fastify/session",
                "value": 7.2,
                "highlight": false
              },
              {
                "label": "h3js/h3",
                "value": 12.5,
                "highlight": false
              },
              {
                "label": "Kludex/uvicorn",
                "value": 22.7,
                "highlight": false
              }
            ]
          },
          {
            "name": "Uncapped, build 236c0d3f",
            "points": [
              {
                "label": "fastify/session",
                "value": 11.2,
                "highlight": true
              },
              {
                "label": "h3js/h3",
                "value": 9.2,
                "highlight": true
              },
              {
                "label": "Kludex/uvicorn",
                "value": 19.8,
                "highlight": true
              }
            ]
          }
        ],
        "note": "Onboarding included. uvicorn needed more than the old 20 minute cap once the caps were removed.",
        "sourceIds": [
          "agent-coding-calibration"
        ]
      },
      {
        "id": "calibration-guardrail-refusals",
        "title": "Guardrail refusals per slice",
        "subtitle": "Tool calls the platform turned back (plan schema, delivery contract, outgoing text, loops)",
        "kind": "bar",
        "unit": "count",
        "yLabel": "Refusals (3 tasks)",
        "series": [
          {
            "name": "Refusals",
            "points": [
              {
                "label": "Baseline (capped)",
                "value": 26,
                "highlight": false
              },
              {
                "label": "Fix wave 1 (capped)",
                "value": 28,
                "highlight": false
              },
              {
                "label": "Uncapped, build f0ac3a8a",
                "value": 25,
                "highlight": false
              },
              {
                "label": "Uncapped, build 236c0d3f",
                "value": 19,
                "highlight": true
              }
            ]
          }
        ],
        "note": "A refusal is a guardrail working, not always a failure: some catch real problems, some were platform defects that the next build fixed.",
        "sourceIds": [
          "agent-coding-calibration"
        ]
      }
    ],
    "tables": [
      {
        "id": "calibration-every-attempt",
        "title": "Every attempt, failures included",
        "columns": [
          {
            "key": "task",
            "label": "Task",
            "unit": "text"
          },
          {
            "key": "slice",
            "label": "Slice",
            "unit": "text"
          },
          {
            "key": "functional",
            "label": "Functional gates",
            "unit": "text"
          },
          {
            "key": "verified",
            "label": "Verified delivery",
            "unit": "text"
          },
          {
            "key": "stop",
            "label": "How it ended",
            "unit": "text"
          },
          {
            "key": "minutes",
            "label": "Minutes",
            "unit": "minutes"
          },
          {
            "key": "costUsd",
            "label": "Cost (notional)",
            "unit": "usd"
          },
          {
            "key": "calls",
            "label": "Calls",
            "unit": "calls"
          },
          {
            "key": "why",
            "label": "Why not verified",
            "unit": "text"
          }
        ],
        "rows": [
          {
            "task": "fastify/session",
            "slice": "Baseline (capped)",
            "functional": "verified",
            "verified": "no",
            "stop": "Escalated to a person",
            "minutes": 9,
            "costUsd": 3.43,
            "calls": 43,
            "why": "run-not-completed"
          },
          {
            "task": "fastify/session",
            "slice": "Fix wave 1 (capped)",
            "functional": "verified",
            "verified": "no",
            "stop": "Hit the $5 cap",
            "minutes": 15.5,
            "costUsd": 4.85,
            "calls": 73,
            "why": "integrity-caveat:cost-capped, run-not-completed"
          },
          {
            "task": "fastify/session",
            "slice": "Uncapped, build f0ac3a8a",
            "functional": "failed",
            "verified": "no",
            "stop": "Escalated to a person",
            "minutes": 7.2,
            "costUsd": 2.44,
            "calls": 29,
            "why": "account-route-mismatch, run-not-completed, acceptance-failed, cross-test-command-failed:reference:default, cross-test-failed:reference:default, empty-patch"
          },
          {
            "task": "fastify/session",
            "slice": "Uncapped, build 236c0d3f",
            "functional": "verified",
            "verified": "yes",
            "stop": "Delivered (in review)",
            "minutes": 11.2,
            "costUsd": 3.53,
            "calls": 52,
            "why": ""
          },
          {
            "task": "h3js/h3",
            "slice": "Baseline (capped)",
            "functional": "verified",
            "verified": "no",
            "stop": "Escalated to a person",
            "minutes": 11.8,
            "costUsd": 4,
            "calls": 53,
            "why": "run-not-completed"
          },
          {
            "task": "h3js/h3",
            "slice": "Fix wave 1 (capped)",
            "functional": "verified",
            "verified": "no",
            "stop": "Escalated to a person",
            "minutes": 13.5,
            "costUsd": 3.29,
            "calls": 46,
            "why": "run-not-completed"
          },
          {
            "task": "h3js/h3",
            "slice": "Uncapped, build f0ac3a8a",
            "functional": "verified",
            "verified": "no",
            "stop": "Delivered (in review)",
            "minutes": 12.5,
            "costUsd": 3.81,
            "calls": 48,
            "why": "account-route-mismatch"
          },
          {
            "task": "h3js/h3",
            "slice": "Uncapped, build 236c0d3f",
            "functional": "failed",
            "verified": "no",
            "stop": "Escalated to a person",
            "minutes": 9.2,
            "costUsd": 2.96,
            "calls": 35,
            "why": "run-not-completed, acceptance-failed, cross-test-command-failed:reference:default, cross-test-failed:reference:default, empty-patch"
          },
          {
            "task": "Kludex/uvicorn",
            "slice": "Baseline (capped)",
            "functional": "unverified",
            "verified": "no",
            "stop": "Hit the $5 cap",
            "minutes": 17.9,
            "costUsd": 4.88,
            "calls": 67,
            "why": "integrity-caveat:cost-capped, run-not-completed, acceptance-missing, base-acceptance-not-run, base-passToPass-not-passed, candidate-acceptance-not-run, cross-tests-not-run:candidate:locked, cross-tests-not-run:candidate:ws17, cross-tests-not-run:reference:locked, cross-tests-not-run:reference:ws17, step-fail:install, step-skipped:build, step-skipped:lint, step-skipped:test, step-skipped:typecheck"
          },
          {
            "task": "Kludex/uvicorn",
            "slice": "Fix wave 1 (capped)",
            "functional": "unverified",
            "verified": "no",
            "stop": "Hit the 20 min cap",
            "minutes": 20.3,
            "costUsd": 4.63,
            "calls": 68,
            "why": "run-not-completed, acceptance-missing, base-acceptance-not-run, base-passToPass-not-passed, candidate-acceptance-not-run, cross-tests-not-run:candidate:locked, cross-tests-not-run:candidate:ws17, cross-tests-not-run:reference:locked, cross-tests-not-run:reference:ws17, step-fail:install, step-skipped:build, step-skipped:lint, step-skipped:test, step-skipped:typecheck"
          },
          {
            "task": "Kludex/uvicorn",
            "slice": "Uncapped, build f0ac3a8a",
            "functional": "unverified",
            "verified": "no",
            "stop": "Delivered (in review)",
            "minutes": 22.7,
            "costUsd": 4.74,
            "calls": 73,
            "why": "account-route-mismatch, acceptance-missing, base-acceptance-not-run, base-passToPass-not-passed, candidate-acceptance-not-run, cross-tests-not-run:candidate:ws17, cross-tests-not-run:reference:ws17"
          },
          {
            "task": "Kludex/uvicorn",
            "slice": "Uncapped, build 236c0d3f",
            "functional": "unverified",
            "verified": "no",
            "stop": "Delivered (in review)",
            "minutes": 19.8,
            "costUsd": 4.57,
            "calls": 62,
            "why": "base-did-not-fail, cross-test-command-failed:candidate:ws17"
          }
        ]
      }
    ]
  },
  "sources": [
    {
      "id": "agent-coding-calibration",
      "title": "Coding calibration: fastify/session, h3, uvicorn",
      "kind": "run",
      "date": "2026-10-05",
      "note": "Three real upstream tasks, one attempt per task per platform slice, offline gates against the merged reference.",
      "data": [
        "/benchmarks/raw/calibration/slices.json"
      ]
    }
  ]
}
