{
 "note": "Receipts copied from public-runs/provider-h2h. Every attempt is kept, failures included.",
 "cases": [
  {
   "id": "fix-median",
   "title": "Fix a buggy median function",
   "kind": "code-fix",
   "validator": "sandbox"
  },
  {
   "id": "invoice-json",
   "title": "Extract invoice fields to JSON",
   "kind": "extraction",
   "validator": "json"
  },
  {
   "id": "shift-arithmetic",
   "title": "Multi-step shift arithmetic",
   "kind": "reasoning",
   "validator": "exact",
   "expected": "2292"
  },
  {
   "id": "flatten-iterative",
   "title": "Refactor recursion to iteration",
   "kind": "refactor",
   "validator": "sandbox"
  },
  {
   "id": "ticket-triage",
   "title": "Classify six support tickets",
   "kind": "classification",
   "validator": "json"
  }
 ],
 "batches": [
  {
   "batch": "batch-1",
   "schema": "agent-provider-h2h@1",
   "route": "claude-cli",
   "startedAt": "2026-10-05T20:36:30.185Z",
   "finishedAt": "2026-10-05T20:43:55.273Z",
   "stopReason": null,
   "trimmed": [],
   "generatedAt": null
  },
  {
   "batch": "batch-2",
   "schema": "agent-provider-h2h@1",
   "route": "codex-cli",
   "startedAt": "2026-10-05T20:36:30.184Z",
   "finishedAt": "2026-10-05T20:40:40.352Z",
   "stopReason": null,
   "trimmed": [],
   "generatedAt": null
  },
  {
   "batch": "batch-3",
   "schema": "agent-provider-h2h@1",
   "route": "codex-cli",
   "startedAt": "2026-10-06T13:15:38.084Z",
   "finishedAt": "2026-10-06T13:17:14.070Z",
   "stopReason": null,
   "trimmed": [],
   "generatedAt": null
  }
 ],
 "receipts": [
  {
   "batch": "batch-1",
   "caseId": "fix-median",
   "caseTitle": "Fix a buggy median function",
   "caseKind": "code-fix",
   "route": "claude-cli",
   "modelRequested": "haiku",
   "modelReported": "claude-haiku-4-5-20251001",
   "effort": "default",
   "rep": 1,
   "startedAt": "2026-10-05T20:36:30.185Z",
   "finishedAt": "2026-10-05T20:36:40.900Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 8876,
   "totalMs": 9571,
   "modelTurnMs": 8547,
   "inputTokens": 3761,
   "outputTokens": 1343,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 1242,
   "passed": true,
   "checks": 8,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 235,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "fix-median",
   "caseTitle": "Fix a buggy median function",
   "caseKind": "code-fix",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 1,
   "startedAt": "2026-10-05T20:36:40.901Z",
   "finishedAt": "2026-10-05T20:36:43.826Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 994,
   "totalMs": 2227,
   "modelTurnMs": 1248,
   "inputTokens": 2046,
   "outputTokens": 109,
   "cacheReadTokens": 531,
   "cacheWriteTokens": 1513,
   "reasoningTokens": 0,
   "passed": true,
   "checks": 8,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 232,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "fix-median",
   "caseTitle": "Fix a buggy median function",
   "caseKind": "code-fix",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "default",
   "rep": 1,
   "startedAt": "2026-10-05T20:36:43.826Z",
   "finishedAt": "2026-10-05T20:36:47.241Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 1581,
   "totalMs": 2623,
   "modelTurnMs": 1628,
   "inputTokens": 2040,
   "outputTokens": 102,
   "cacheReadTokens": 531,
   "cacheWriteTokens": 1507,
   "reasoningTokens": 0,
   "passed": true,
   "checks": 8,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 214,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "fix-median",
   "caseTitle": "Fix a buggy median function",
   "caseKind": "code-fix",
   "route": "claude-cli",
   "modelRequested": "fable",
   "modelReported": "claude-fable-5-1",
   "effort": "default",
   "rep": 1,
   "startedAt": "2026-10-05T20:36:47.241Z",
   "finishedAt": "2026-10-05T20:36:50.944Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 1705,
   "totalMs": 2749,
   "modelTurnMs": 1706,
   "inputTokens": 3192,
   "outputTokens": 102,
   "cacheReadTokens": 531,
   "cacheWriteTokens": 2659,
   "reasoningTokens": 0,
   "passed": true,
   "checks": 8,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 214,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "fix-median",
   "caseTitle": "Fix a buggy median function",
   "caseKind": "code-fix",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "low",
   "rep": 1,
   "startedAt": "2026-10-05T20:36:50.944Z",
   "finishedAt": "2026-10-05T20:36:56.882Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 3190,
   "totalMs": 4667,
   "modelTurnMs": 3470,
   "inputTokens": 2039,
   "outputTokens": 102,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 574,
   "reasoningTokens": 0,
   "passed": true,
   "checks": 8,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 214,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "fix-median",
   "caseTitle": "Fix a buggy median function",
   "caseKind": "code-fix",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "high",
   "rep": 1,
   "startedAt": "2026-10-05T20:36:56.882Z",
   "finishedAt": "2026-10-05T20:37:00.466Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 1535,
   "totalMs": 2637,
   "modelTurnMs": 1537,
   "inputTokens": 2040,
   "outputTokens": 102,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 575,
   "reasoningTokens": 0,
   "passed": true,
   "checks": 8,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 214,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "invoice-json",
   "caseTitle": "Extract invoice fields to JSON",
   "caseKind": "extraction",
   "route": "claude-cli",
   "modelRequested": "haiku",
   "modelReported": "claude-haiku-4-5-20251001",
   "effort": "default",
   "rep": 1,
   "startedAt": "2026-10-05T20:37:00.467Z",
   "finishedAt": "2026-10-05T20:37:04.800Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 3143,
   "totalMs": 3968,
   "modelTurnMs": 2953,
   "inputTokens": 3822,
   "outputTokens": 290,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 212,
   "passed": true,
   "checks": 2,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 153,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "invoice-json",
   "caseTitle": "Extract invoice fields to JSON",
   "caseKind": "extraction",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 1,
   "startedAt": "2026-10-05T20:37:04.800Z",
   "finishedAt": "2026-10-05T20:37:07.421Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 1660,
   "totalMs": 2309,
   "modelTurnMs": 1384,
   "inputTokens": 2131,
   "outputTokens": 59,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 666,
   "reasoningTokens": 0,
   "passed": true,
   "checks": 2,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 116,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "invoice-json",
   "caseTitle": "Extract invoice fields to JSON",
   "caseKind": "extraction",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "default",
   "rep": 1,
   "startedAt": "2026-10-05T20:37:07.422Z",
   "finishedAt": "2026-10-05T20:37:10.467Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 1921,
   "totalMs": 2745,
   "modelTurnMs": 1796,
   "inputTokens": 2125,
   "outputTokens": 64,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 660,
   "reasoningTokens": 0,
   "passed": true,
   "checks": 2,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 127,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "invoice-json",
   "caseTitle": "Extract invoice fields to JSON",
   "caseKind": "extraction",
   "route": "claude-cli",
   "modelRequested": "fable",
   "modelReported": "claude-fable-5-1",
   "effort": "default",
   "rep": 1,
   "startedAt": "2026-10-05T20:37:10.468Z",
   "finishedAt": "2026-10-05T20:37:12.864Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 1328,
   "totalMs": 2107,
   "modelTurnMs": 1093,
   "inputTokens": 3279,
   "outputTokens": 64,
   "cacheReadTokens": 2638,
   "cacheWriteTokens": 639,
   "reasoningTokens": 0,
   "passed": true,
   "checks": 2,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 127,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "invoice-json",
   "caseTitle": "Extract invoice fields to JSON",
   "caseKind": "extraction",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "low",
   "rep": 1,
   "startedAt": "2026-10-05T20:37:12.865Z",
   "finishedAt": "2026-10-05T20:37:15.893Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 1942,
   "totalMs": 2725,
   "modelTurnMs": 1755,
   "inputTokens": 2128,
   "outputTokens": 64,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 663,
   "reasoningTokens": 0,
   "passed": true,
   "checks": 2,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 127,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "invoice-json",
   "caseTitle": "Extract invoice fields to JSON",
   "caseKind": "extraction",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "high",
   "rep": 1,
   "startedAt": "2026-10-05T20:37:15.894Z",
   "finishedAt": "2026-10-05T20:37:19.008Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 2011,
   "totalMs": 2817,
   "modelTurnMs": 1838,
   "inputTokens": 2127,
   "outputTokens": 64,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 662,
   "reasoningTokens": 0,
   "passed": true,
   "checks": 2,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 127,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "shift-arithmetic",
   "caseTitle": "Multi-step shift arithmetic",
   "caseKind": "reasoning",
   "route": "claude-cli",
   "modelRequested": "haiku",
   "modelReported": "claude-haiku-4-5-20251001",
   "effort": "default",
   "rep": 1,
   "startedAt": "2026-10-05T20:37:19.008Z",
   "finishedAt": "2026-10-05T20:37:22.464Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 2781,
   "totalMs": 3164,
   "modelTurnMs": 2239,
   "inputTokens": 3753,
   "outputTokens": 275,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 268,
   "passed": true,
   "checks": 1,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 4,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "shift-arithmetic",
   "caseTitle": "Multi-step shift arithmetic",
   "caseKind": "reasoning",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 1,
   "startedAt": "2026-10-05T20:37:22.464Z",
   "finishedAt": "2026-10-05T20:37:25.403Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 1454,
   "totalMs": 2642,
   "modelTurnMs": 1373,
   "inputTokens": 2022,
   "outputTokens": 107,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 557,
   "reasoningTokens": 0,
   "passed": false,
   "checks": 1,
   "failures": [
    "Output does not exactly match the expected text."
   ],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 168,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS."
   ],
   "failedOutputLastLine": "2292",
   "failedOutputLineCount": 6
  },
  {
   "batch": "batch-1",
   "caseId": "shift-arithmetic",
   "caseTitle": "Multi-step shift arithmetic",
   "caseKind": "reasoning",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "default",
   "rep": 1,
   "startedAt": "2026-10-05T20:37:25.403Z",
   "finishedAt": "2026-10-05T20:37:28.977Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 2910,
   "totalMs": 3253,
   "modelTurnMs": 2035,
   "inputTokens": 2017,
   "outputTokens": 60,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 552,
   "reasoningTokens": 56,
   "passed": true,
   "checks": 1,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 4,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "shift-arithmetic",
   "caseTitle": "Multi-step shift arithmetic",
   "caseKind": "reasoning",
   "route": "claude-cli",
   "modelRequested": "fable",
   "modelReported": "claude-fable-5-1",
   "effort": "default",
   "rep": 1,
   "startedAt": "2026-10-05T20:37:28.977Z",
   "finishedAt": "2026-10-05T20:37:30.855Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 1143,
   "totalMs": 1564,
   "modelTurnMs": 603,
   "inputTokens": 3170,
   "outputTokens": 4,
   "cacheReadTokens": 2638,
   "cacheWriteTokens": 530,
   "reasoningTokens": 0,
   "passed": true,
   "checks": 1,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 4,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "shift-arithmetic",
   "caseTitle": "Multi-step shift arithmetic",
   "caseKind": "reasoning",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "low",
   "rep": 1,
   "startedAt": "2026-10-05T20:37:30.856Z",
   "finishedAt": "2026-10-05T20:37:34.006Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 2386,
   "totalMs": 2836,
   "modelTurnMs": 1879,
   "inputTokens": 2017,
   "outputTokens": 55,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 552,
   "reasoningTokens": 51,
   "passed": true,
   "checks": 1,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 4,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "shift-arithmetic",
   "caseTitle": "Multi-step shift arithmetic",
   "caseKind": "reasoning",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "high",
   "rep": 1,
   "startedAt": "2026-10-05T20:37:34.007Z",
   "finishedAt": "2026-10-05T20:37:37.018Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 2305,
   "totalMs": 2708,
   "modelTurnMs": 1731,
   "inputTokens": 2018,
   "outputTokens": 60,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 553,
   "reasoningTokens": 56,
   "passed": true,
   "checks": 1,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 4,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "flatten-iterative",
   "caseTitle": "Refactor recursion to iteration",
   "caseKind": "refactor",
   "route": "claude-cli",
   "modelRequested": "haiku",
   "modelReported": "claude-haiku-4-5-20251001",
   "effort": "default",
   "rep": 1,
   "startedAt": "2026-10-05T20:37:37.019Z",
   "finishedAt": "2026-10-05T20:38:01.309Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 22274,
   "totalMs": 23569,
   "modelTurnMs": 22779,
   "inputTokens": 3788,
   "outputTokens": 2812,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 2633,
   "passed": true,
   "checks": 10,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 477,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "flatten-iterative",
   "caseTitle": "Refactor recursion to iteration",
   "caseKind": "refactor",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 1,
   "startedAt": "2026-10-05T20:38:01.309Z",
   "finishedAt": "2026-10-05T20:38:08.365Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 5137,
   "totalMs": 6319,
   "modelTurnMs": 5434,
   "inputTokens": 2081,
   "outputTokens": 692,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 616,
   "reasoningTokens": 451,
   "passed": true,
   "checks": 10,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 589,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "flatten-iterative",
   "caseTitle": "Refactor recursion to iteration",
   "caseKind": "refactor",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "default",
   "rep": 1,
   "startedAt": "2026-10-05T20:38:08.366Z",
   "finishedAt": "2026-10-05T20:38:18.043Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 7231,
   "totalMs": 8914,
   "modelTurnMs": 7891,
   "inputTokens": 2076,
   "outputTokens": 830,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 611,
   "reasoningTokens": 598,
   "passed": true,
   "checks": 10,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 557,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "flatten-iterative",
   "caseTitle": "Refactor recursion to iteration",
   "caseKind": "refactor",
   "route": "claude-cli",
   "modelRequested": "fable",
   "modelReported": "claude-fable-5-1",
   "effort": "default",
   "rep": 1,
   "startedAt": "2026-10-05T20:38:18.043Z",
   "finishedAt": "2026-10-05T20:38:26.088Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 5437,
   "totalMs": 7294,
   "modelTurnMs": 6417,
   "inputTokens": 3228,
   "outputTokens": 725,
   "cacheReadTokens": 2638,
   "cacheWriteTokens": 588,
   "reasoningTokens": 494,
   "passed": true,
   "checks": 10,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 555,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "flatten-iterative",
   "caseTitle": "Refactor recursion to iteration",
   "caseKind": "refactor",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "low",
   "rep": 1,
   "startedAt": "2026-10-05T20:38:26.088Z",
   "finishedAt": "2026-10-05T20:38:32.862Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 4249,
   "totalMs": 6034,
   "modelTurnMs": 5226,
   "inputTokens": 2077,
   "outputTokens": 450,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 612,
   "reasoningTokens": 210,
   "passed": true,
   "checks": 10,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 580,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "flatten-iterative",
   "caseTitle": "Refactor recursion to iteration",
   "caseKind": "refactor",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "high",
   "rep": 1,
   "startedAt": "2026-10-05T20:38:32.863Z",
   "finishedAt": "2026-10-05T20:38:45.386Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 9940,
   "totalMs": 11780,
   "modelTurnMs": 10906,
   "inputTokens": 2077,
   "outputTokens": 1059,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 612,
   "reasoningTokens": 819,
   "passed": true,
   "checks": 10,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 580,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "ticket-triage",
   "caseTitle": "Classify six support tickets",
   "caseKind": "classification",
   "route": "claude-cli",
   "modelRequested": "haiku",
   "modelReported": "claude-haiku-4-5-20251001",
   "effort": "default",
   "rep": 1,
   "startedAt": "2026-10-05T20:38:45.386Z",
   "finishedAt": "2026-10-05T20:38:49.619Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 3341,
   "totalMs": 3908,
   "modelTurnMs": 2761,
   "inputTokens": 3828,
   "outputTokens": 303,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 261,
   "passed": true,
   "checks": 2,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 97,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "ticket-triage",
   "caseTitle": "Classify six support tickets",
   "caseKind": "classification",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 1,
   "startedAt": "2026-10-05T20:38:49.620Z",
   "finishedAt": "2026-10-05T20:38:52.128Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 1198,
   "totalMs": 2210,
   "modelTurnMs": 1239,
   "inputTokens": 2149,
   "outputTokens": 44,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 684,
   "reasoningTokens": 0,
   "passed": true,
   "checks": 2,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 85,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "ticket-triage",
   "caseTitle": "Classify six support tickets",
   "caseKind": "classification",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "default",
   "rep": 1,
   "startedAt": "2026-10-05T20:38:52.128Z",
   "finishedAt": "2026-10-05T20:38:54.941Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 1875,
   "totalMs": 2506,
   "modelTurnMs": 1570,
   "inputTokens": 2145,
   "outputTokens": 44,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 680,
   "reasoningTokens": 0,
   "passed": true,
   "checks": 2,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 85,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "ticket-triage",
   "caseTitle": "Classify six support tickets",
   "caseKind": "classification",
   "route": "claude-cli",
   "modelRequested": "fable",
   "modelReported": "claude-fable-5-1",
   "effort": "default",
   "rep": 1,
   "startedAt": "2026-10-05T20:38:54.941Z",
   "finishedAt": "2026-10-05T20:38:57.031Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 1196,
   "totalMs": 1789,
   "modelTurnMs": 770,
   "inputTokens": 3297,
   "outputTokens": 44,
   "cacheReadTokens": 2638,
   "cacheWriteTokens": 657,
   "reasoningTokens": 0,
   "passed": true,
   "checks": 2,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 85,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "ticket-triage",
   "caseTitle": "Classify six support tickets",
   "caseKind": "classification",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "low",
   "rep": 1,
   "startedAt": "2026-10-05T20:38:57.032Z",
   "finishedAt": "2026-10-05T20:38:59.918Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 1997,
   "totalMs": 2578,
   "modelTurnMs": 1641,
   "inputTokens": 2145,
   "outputTokens": 44,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 680,
   "reasoningTokens": 0,
   "passed": true,
   "checks": 2,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 85,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "ticket-triage",
   "caseTitle": "Classify six support tickets",
   "caseKind": "classification",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "high",
   "rep": 1,
   "startedAt": "2026-10-05T20:38:59.918Z",
   "finishedAt": "2026-10-05T20:39:02.891Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 2000,
   "totalMs": 2669,
   "modelTurnMs": 1693,
   "inputTokens": 2145,
   "outputTokens": 84,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 680,
   "reasoningTokens": 40,
   "passed": true,
   "checks": 2,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 85,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "fix-median",
   "caseTitle": "Fix a buggy median function",
   "caseKind": "code-fix",
   "route": "claude-cli",
   "modelRequested": "haiku",
   "modelReported": "claude-haiku-4-5-20251001",
   "effort": "default",
   "rep": 2,
   "startedAt": "2026-10-05T20:39:02.891Z",
   "finishedAt": "2026-10-05T20:39:10.443Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 5922,
   "totalMs": 6809,
   "modelTurnMs": 6019,
   "inputTokens": 3759,
   "outputTokens": 671,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 567,
   "passed": true,
   "checks": 8,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 238,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "fix-median",
   "caseTitle": "Fix a buggy median function",
   "caseKind": "code-fix",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 2,
   "startedAt": "2026-10-05T20:39:10.443Z",
   "finishedAt": "2026-10-05T20:39:13.816Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 1561,
   "totalMs": 2670,
   "modelTurnMs": 1313,
   "inputTokens": 2044,
   "outputTokens": 109,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 579,
   "reasoningTokens": 0,
   "passed": true,
   "checks": 8,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 232,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "fix-median",
   "caseTitle": "Fix a buggy median function",
   "caseKind": "code-fix",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "default",
   "rep": 2,
   "startedAt": "2026-10-05T20:39:13.816Z",
   "finishedAt": "2026-10-05T20:39:18.861Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 3202,
   "totalMs": 4281,
   "modelTurnMs": 3321,
   "inputTokens": 2039,
   "outputTokens": 102,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 574,
   "reasoningTokens": 0,
   "passed": true,
   "checks": 8,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 214,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "fix-median",
   "caseTitle": "Fix a buggy median function",
   "caseKind": "code-fix",
   "route": "claude-cli",
   "modelRequested": "fable",
   "modelReported": "claude-fable-5-1",
   "effort": "default",
   "rep": 2,
   "startedAt": "2026-10-05T20:39:18.861Z",
   "finishedAt": "2026-10-05T20:39:22.153Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 1580,
   "totalMs": 2605,
   "modelTurnMs": 1673,
   "inputTokens": 3191,
   "outputTokens": 102,
   "cacheReadTokens": 2990,
   "cacheWriteTokens": 199,
   "reasoningTokens": 0,
   "passed": true,
   "checks": 8,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 214,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "fix-median",
   "caseTitle": "Fix a buggy median function",
   "caseKind": "code-fix",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "low",
   "rep": 2,
   "startedAt": "2026-10-05T20:39:22.153Z",
   "finishedAt": "2026-10-05T20:39:25.467Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 1583,
   "totalMs": 2610,
   "modelTurnMs": 1558,
   "inputTokens": 2040,
   "outputTokens": 102,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 575,
   "reasoningTokens": 0,
   "passed": true,
   "checks": 8,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 214,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "fix-median",
   "caseTitle": "Fix a buggy median function",
   "caseKind": "code-fix",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "high",
   "rep": 2,
   "startedAt": "2026-10-05T20:39:25.468Z",
   "finishedAt": "2026-10-05T20:39:29.021Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 1551,
   "totalMs": 2852,
   "modelTurnMs": 1663,
   "inputTokens": 2040,
   "outputTokens": 102,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 575,
   "reasoningTokens": 0,
   "passed": true,
   "checks": 8,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 214,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "invoice-json",
   "caseTitle": "Extract invoice fields to JSON",
   "caseKind": "extraction",
   "route": "claude-cli",
   "modelRequested": "haiku",
   "modelReported": "claude-haiku-4-5-20251001",
   "effort": "default",
   "rep": 2,
   "startedAt": "2026-10-05T20:39:29.022Z",
   "finishedAt": "2026-10-05T20:39:33.800Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 3590,
   "totalMs": 4475,
   "modelTurnMs": 3469,
   "inputTokens": 3822,
   "outputTokens": 367,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 290,
   "passed": true,
   "checks": 2,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 153,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "invoice-json",
   "caseTitle": "Extract invoice fields to JSON",
   "caseKind": "extraction",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 2,
   "startedAt": "2026-10-05T20:39:33.800Z",
   "finishedAt": "2026-10-05T20:39:37.022Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 2008,
   "totalMs": 2919,
   "modelTurnMs": 1660,
   "inputTokens": 2132,
   "outputTokens": 64,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 667,
   "reasoningTokens": 0,
   "passed": true,
   "checks": 2,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 127,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "invoice-json",
   "caseTitle": "Extract invoice fields to JSON",
   "caseKind": "extraction",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "default",
   "rep": 2,
   "startedAt": "2026-10-05T20:39:37.023Z",
   "finishedAt": "2026-10-05T20:39:40.078Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 1875,
   "totalMs": 2737,
   "modelTurnMs": 1774,
   "inputTokens": 2127,
   "outputTokens": 64,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 662,
   "reasoningTokens": 0,
   "passed": true,
   "checks": 2,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 127,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "invoice-json",
   "caseTitle": "Extract invoice fields to JSON",
   "caseKind": "extraction",
   "route": "claude-cli",
   "modelRequested": "fable",
   "modelReported": "claude-fable-5-1",
   "effort": "default",
   "rep": 2,
   "startedAt": "2026-10-05T20:39:40.079Z",
   "finishedAt": "2026-10-05T20:39:42.319Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 1156,
   "totalMs": 1941,
   "modelTurnMs": 939,
   "inputTokens": 3280,
   "outputTokens": 64,
   "cacheReadTokens": 3077,
   "cacheWriteTokens": 201,
   "reasoningTokens": 0,
   "passed": true,
   "checks": 2,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 127,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "invoice-json",
   "caseTitle": "Extract invoice fields to JSON",
   "caseKind": "extraction",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "low",
   "rep": 2,
   "startedAt": "2026-10-05T20:39:42.319Z",
   "finishedAt": "2026-10-05T20:39:46.944Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 3484,
   "totalMs": 4317,
   "modelTurnMs": 3350,
   "inputTokens": 2126,
   "outputTokens": 64,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 661,
   "reasoningTokens": 0,
   "passed": true,
   "checks": 2,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 127,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "invoice-json",
   "caseTitle": "Extract invoice fields to JSON",
   "caseKind": "extraction",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "high",
   "rep": 2,
   "startedAt": "2026-10-05T20:39:46.945Z",
   "finishedAt": "2026-10-05T20:39:49.927Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 1839,
   "totalMs": 2637,
   "modelTurnMs": 1702,
   "inputTokens": 2127,
   "outputTokens": 64,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 662,
   "reasoningTokens": 0,
   "passed": true,
   "checks": 2,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 127,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "shift-arithmetic",
   "caseTitle": "Multi-step shift arithmetic",
   "caseKind": "reasoning",
   "route": "claude-cli",
   "modelRequested": "haiku",
   "modelReported": "claude-haiku-4-5-20251001",
   "effort": "default",
   "rep": 2,
   "startedAt": "2026-10-05T20:39:49.927Z",
   "finishedAt": "2026-10-05T20:39:53.760Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 3056,
   "totalMs": 3603,
   "modelTurnMs": 2485,
   "inputTokens": 3754,
   "outputTokens": 293,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 285,
   "passed": true,
   "checks": 1,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 4,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "shift-arithmetic",
   "caseTitle": "Multi-step shift arithmetic",
   "caseKind": "reasoning",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 2,
   "startedAt": "2026-10-05T20:39:53.761Z",
   "finishedAt": "2026-10-05T20:39:56.260Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 1017,
   "totalMs": 2254,
   "modelTurnMs": 1284,
   "inputTokens": 2021,
   "outputTokens": 107,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 556,
   "reasoningTokens": 0,
   "passed": false,
   "checks": 1,
   "failures": [
    "Output does not exactly match the expected text."
   ],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 170,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS."
   ],
   "failedOutputLastLine": "2292",
   "failedOutputLineCount": 6
  },
  {
   "batch": "batch-1",
   "caseId": "shift-arithmetic",
   "caseTitle": "Multi-step shift arithmetic",
   "caseKind": "reasoning",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "default",
   "rep": 2,
   "startedAt": "2026-10-05T20:39:56.261Z",
   "finishedAt": "2026-10-05T20:39:59.981Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 3120,
   "totalMs": 3486,
   "modelTurnMs": 2630,
   "inputTokens": 2018,
   "outputTokens": 60,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 553,
   "reasoningTokens": 56,
   "passed": true,
   "checks": 1,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 4,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "shift-arithmetic",
   "caseTitle": "Multi-step shift arithmetic",
   "caseKind": "reasoning",
   "route": "claude-cli",
   "modelRequested": "fable",
   "modelReported": "claude-fable-5-1",
   "effort": "default",
   "rep": 2,
   "startedAt": "2026-10-05T20:39:59.981Z",
   "finishedAt": "2026-10-05T20:40:01.659Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 1029,
   "totalMs": 1439,
   "modelTurnMs": 562,
   "inputTokens": 3167,
   "outputTokens": 4,
   "cacheReadTokens": 2968,
   "cacheWriteTokens": 197,
   "reasoningTokens": 0,
   "passed": true,
   "checks": 1,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 4,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "shift-arithmetic",
   "caseTitle": "Multi-step shift arithmetic",
   "caseKind": "reasoning",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "low",
   "rep": 2,
   "startedAt": "2026-10-05T20:40:01.660Z",
   "finishedAt": "2026-10-05T20:40:04.783Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 2482,
   "totalMs": 2889,
   "modelTurnMs": 1988,
   "inputTokens": 2018,
   "outputTokens": 60,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 553,
   "reasoningTokens": 56,
   "passed": true,
   "checks": 1,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 4,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "shift-arithmetic",
   "caseTitle": "Multi-step shift arithmetic",
   "caseKind": "reasoning",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "high",
   "rep": 2,
   "startedAt": "2026-10-05T20:40:04.784Z",
   "finishedAt": "2026-10-05T20:40:07.671Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 2301,
   "totalMs": 2656,
   "modelTurnMs": 1756,
   "inputTokens": 2018,
   "outputTokens": 60,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 553,
   "reasoningTokens": 56,
   "passed": true,
   "checks": 1,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 4,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "flatten-iterative",
   "caseTitle": "Refactor recursion to iteration",
   "caseKind": "refactor",
   "route": "claude-cli",
   "modelRequested": "haiku",
   "modelReported": "claude-haiku-4-5-20251001",
   "effort": "default",
   "rep": 2,
   "startedAt": "2026-10-05T20:40:07.672Z",
   "finishedAt": "2026-10-05T20:40:31.326Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 21570,
   "totalMs": 23106,
   "modelTurnMs": 22321,
   "inputTokens": 3789,
   "outputTokens": 2851,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 2643,
   "passed": true,
   "checks": 10,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 585,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "flatten-iterative",
   "caseTitle": "Refactor recursion to iteration",
   "caseKind": "refactor",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 2,
   "startedAt": "2026-10-05T20:40:31.326Z",
   "finishedAt": "2026-10-05T20:40:37.730Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 4620,
   "totalMs": 5895,
   "modelTurnMs": 5190,
   "inputTokens": 2081,
   "outputTokens": 574,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 616,
   "reasoningTokens": 330,
   "passed": true,
   "checks": 10,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 602,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "flatten-iterative",
   "caseTitle": "Refactor recursion to iteration",
   "caseKind": "refactor",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "default",
   "rep": 2,
   "startedAt": "2026-10-05T20:40:37.730Z",
   "finishedAt": "2026-10-05T20:40:46.438Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 6480,
   "totalMs": 8180,
   "modelTurnMs": 7503,
   "inputTokens": 2077,
   "outputTokens": 787,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 612,
   "reasoningTokens": 557,
   "passed": true,
   "checks": 10,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 562,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "flatten-iterative",
   "caseTitle": "Refactor recursion to iteration",
   "caseKind": "refactor",
   "route": "claude-cli",
   "modelRequested": "fable",
   "modelReported": "claude-fable-5-1",
   "effort": "default",
   "rep": 2,
   "startedAt": "2026-10-05T20:40:46.438Z",
   "finishedAt": "2026-10-05T20:40:56.541Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 7801,
   "totalMs": 9560,
   "modelTurnMs": 8510,
   "inputTokens": 3229,
   "outputTokens": 847,
   "cacheReadTokens": 3027,
   "cacheWriteTokens": 200,
   "reasoningTokens": 621,
   "passed": true,
   "checks": 10,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 548,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "flatten-iterative",
   "caseTitle": "Refactor recursion to iteration",
   "caseKind": "refactor",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "low",
   "rep": 2,
   "startedAt": "2026-10-05T20:40:56.541Z",
   "finishedAt": "2026-10-05T20:41:02.312Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 3299,
   "totalMs": 5204,
   "modelTurnMs": 4289,
   "inputTokens": 2077,
   "outputTokens": 402,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 612,
   "reasoningTokens": 162,
   "passed": true,
   "checks": 10,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 580,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "flatten-iterative",
   "caseTitle": "Refactor recursion to iteration",
   "caseKind": "refactor",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "high",
   "rep": 2,
   "startedAt": "2026-10-05T20:41:02.312Z",
   "finishedAt": "2026-10-05T20:41:14.138Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 9524,
   "totalMs": 11273,
   "modelTurnMs": 10533,
   "inputTokens": 2077,
   "outputTokens": 1094,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 612,
   "reasoningTokens": 857,
   "passed": true,
   "checks": 10,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 574,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "ticket-triage",
   "caseTitle": "Classify six support tickets",
   "caseKind": "classification",
   "route": "claude-cli",
   "modelRequested": "haiku",
   "modelReported": "claude-haiku-4-5-20251001",
   "effort": "default",
   "rep": 2,
   "startedAt": "2026-10-05T20:41:14.139Z",
   "finishedAt": "2026-10-05T20:41:18.240Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 3261,
   "totalMs": 3857,
   "modelTurnMs": 2955,
   "inputTokens": 3826,
   "outputTokens": 343,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 302,
   "passed": true,
   "checks": 2,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 97,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "ticket-triage",
   "caseTitle": "Classify six support tickets",
   "caseKind": "classification",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 2,
   "startedAt": "2026-10-05T20:41:18.240Z",
   "finishedAt": "2026-10-05T20:41:20.657Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 1318,
   "totalMs": 2173,
   "modelTurnMs": 1342,
   "inputTokens": 2149,
   "outputTokens": 44,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 684,
   "reasoningTokens": 0,
   "passed": true,
   "checks": 2,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 85,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "ticket-triage",
   "caseTitle": "Classify six support tickets",
   "caseKind": "classification",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "default",
   "rep": 2,
   "startedAt": "2026-10-05T20:41:20.657Z",
   "finishedAt": "2026-10-05T20:41:23.373Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 1862,
   "totalMs": 2470,
   "modelTurnMs": 1527,
   "inputTokens": 2146,
   "outputTokens": 44,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 681,
   "reasoningTokens": 0,
   "passed": true,
   "checks": 2,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 85,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "ticket-triage",
   "caseTitle": "Classify six support tickets",
   "caseKind": "classification",
   "route": "claude-cli",
   "modelRequested": "fable",
   "modelReported": "claude-fable-5-1",
   "effort": "default",
   "rep": 2,
   "startedAt": "2026-10-05T20:41:23.374Z",
   "finishedAt": "2026-10-05T20:41:25.247Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 1032,
   "totalMs": 1630,
   "modelTurnMs": 757,
   "inputTokens": 3296,
   "outputTokens": 44,
   "cacheReadTokens": 3095,
   "cacheWriteTokens": 199,
   "reasoningTokens": 0,
   "passed": true,
   "checks": 2,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 85,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "ticket-triage",
   "caseTitle": "Classify six support tickets",
   "caseKind": "classification",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "low",
   "rep": 2,
   "startedAt": "2026-10-05T20:41:25.248Z",
   "finishedAt": "2026-10-05T20:41:27.839Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 1750,
   "totalMs": 2349,
   "modelTurnMs": 1497,
   "inputTokens": 2146,
   "outputTokens": 44,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 681,
   "reasoningTokens": 0,
   "passed": true,
   "checks": 2,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 85,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "ticket-triage",
   "caseTitle": "Classify six support tickets",
   "caseKind": "classification",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "high",
   "rep": 2,
   "startedAt": "2026-10-05T20:41:27.840Z",
   "finishedAt": "2026-10-05T20:41:30.926Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 2038,
   "totalMs": 2840,
   "modelTurnMs": 1801,
   "inputTokens": 2145,
   "outputTokens": 44,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 680,
   "reasoningTokens": 0,
   "passed": true,
   "checks": 2,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 85,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "fix-median",
   "caseTitle": "Fix a buggy median function",
   "caseKind": "code-fix",
   "route": "claude-cli",
   "modelRequested": "haiku",
   "modelReported": "claude-haiku-4-5-20251001",
   "effort": "default",
   "rep": 3,
   "startedAt": "2026-10-05T20:41:30.926Z",
   "finishedAt": "2026-10-05T20:41:41.341Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 9061,
   "totalMs": 9893,
   "modelTurnMs": 9094,
   "inputTokens": 3761,
   "outputTokens": 1074,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 974,
   "passed": true,
   "checks": 8,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 232,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "fix-median",
   "caseTitle": "Fix a buggy median function",
   "caseKind": "code-fix",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 3,
   "startedAt": "2026-10-05T20:41:41.341Z",
   "finishedAt": "2026-10-05T20:41:44.127Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 1071,
   "totalMs": 2240,
   "modelTurnMs": 1304,
   "inputTokens": 2044,
   "outputTokens": 109,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 579,
   "reasoningTokens": 0,
   "passed": true,
   "checks": 8,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 232,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "fix-median",
   "caseTitle": "Fix a buggy median function",
   "caseKind": "code-fix",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "default",
   "rep": 3,
   "startedAt": "2026-10-05T20:41:44.127Z",
   "finishedAt": "2026-10-05T20:41:47.377Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 1564,
   "totalMs": 2697,
   "modelTurnMs": 1639,
   "inputTokens": 2040,
   "outputTokens": 102,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 575,
   "reasoningTokens": 0,
   "passed": true,
   "checks": 8,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 214,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "fix-median",
   "caseTitle": "Fix a buggy median function",
   "caseKind": "code-fix",
   "route": "claude-cli",
   "modelRequested": "fable",
   "modelReported": "claude-fable-5-1",
   "effort": "default",
   "rep": 3,
   "startedAt": "2026-10-05T20:41:47.378Z",
   "finishedAt": "2026-10-05T20:41:50.511Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 1520,
   "totalMs": 2608,
   "modelTurnMs": 1696,
   "inputTokens": 3192,
   "outputTokens": 102,
   "cacheReadTokens": 2990,
   "cacheWriteTokens": 200,
   "reasoningTokens": 0,
   "passed": true,
   "checks": 8,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 214,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "fix-median",
   "caseTitle": "Fix a buggy median function",
   "caseKind": "code-fix",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "low",
   "rep": 3,
   "startedAt": "2026-10-05T20:41:50.512Z",
   "finishedAt": "2026-10-05T20:41:53.580Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 1454,
   "totalMs": 2539,
   "modelTurnMs": 1606,
   "inputTokens": 2039,
   "outputTokens": 102,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 574,
   "reasoningTokens": 0,
   "passed": true,
   "checks": 8,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 214,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "fix-median",
   "caseTitle": "Fix a buggy median function",
   "caseKind": "code-fix",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "high",
   "rep": 3,
   "startedAt": "2026-10-05T20:41:53.581Z",
   "finishedAt": "2026-10-05T20:41:56.569Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 1397,
   "totalMs": 2453,
   "modelTurnMs": 1601,
   "inputTokens": 2040,
   "outputTokens": 102,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 575,
   "reasoningTokens": 0,
   "passed": true,
   "checks": 8,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 214,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "invoice-json",
   "caseTitle": "Extract invoice fields to JSON",
   "caseKind": "extraction",
   "route": "claude-cli",
   "modelRequested": "haiku",
   "modelReported": "claude-haiku-4-5-20251001",
   "effort": "default",
   "rep": 3,
   "startedAt": "2026-10-05T20:41:56.570Z",
   "finishedAt": "2026-10-05T20:42:01.232Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 3632,
   "totalMs": 4428,
   "modelTurnMs": 3632,
   "inputTokens": 3822,
   "outputTokens": 374,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 297,
   "passed": true,
   "checks": 2,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 153,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "invoice-json",
   "caseTitle": "Extract invoice fields to JSON",
   "caseKind": "extraction",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 3,
   "startedAt": "2026-10-05T20:42:01.232Z",
   "finishedAt": "2026-10-05T20:42:03.718Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 1644,
   "totalMs": 2245,
   "modelTurnMs": 1409,
   "inputTokens": 2131,
   "outputTokens": 64,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 666,
   "reasoningTokens": 0,
   "passed": true,
   "checks": 2,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 127,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "invoice-json",
   "caseTitle": "Extract invoice fields to JSON",
   "caseKind": "extraction",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "default",
   "rep": 3,
   "startedAt": "2026-10-05T20:42:03.719Z",
   "finishedAt": "2026-10-05T20:42:06.551Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 1888,
   "totalMs": 2600,
   "modelTurnMs": 1842,
   "inputTokens": 2127,
   "outputTokens": 64,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 662,
   "reasoningTokens": 0,
   "passed": true,
   "checks": 2,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 127,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "invoice-json",
   "caseTitle": "Extract invoice fields to JSON",
   "caseKind": "extraction",
   "route": "claude-cli",
   "modelRequested": "fable",
   "modelReported": "claude-fable-5-1",
   "effort": "default",
   "rep": 3,
   "startedAt": "2026-10-05T20:42:06.552Z",
   "finishedAt": "2026-10-05T20:42:08.738Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 1042,
   "totalMs": 1944,
   "modelTurnMs": 939,
   "inputTokens": 3278,
   "outputTokens": 64,
   "cacheReadTokens": 3077,
   "cacheWriteTokens": 199,
   "reasoningTokens": 0,
   "passed": true,
   "checks": 2,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 127,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "invoice-json",
   "caseTitle": "Extract invoice fields to JSON",
   "caseKind": "extraction",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "low",
   "rep": 3,
   "startedAt": "2026-10-05T20:42:08.738Z",
   "finishedAt": "2026-10-05T20:42:11.532Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 1736,
   "totalMs": 2565,
   "modelTurnMs": 1688,
   "inputTokens": 2129,
   "outputTokens": 64,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 664,
   "reasoningTokens": 0,
   "passed": true,
   "checks": 2,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 127,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "invoice-json",
   "caseTitle": "Extract invoice fields to JSON",
   "caseKind": "extraction",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "high",
   "rep": 3,
   "startedAt": "2026-10-05T20:42:11.533Z",
   "finishedAt": "2026-10-05T20:42:14.440Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 1889,
   "totalMs": 2674,
   "modelTurnMs": 1820,
   "inputTokens": 2128,
   "outputTokens": 64,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 663,
   "reasoningTokens": 0,
   "passed": true,
   "checks": 2,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 127,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "shift-arithmetic",
   "caseTitle": "Multi-step shift arithmetic",
   "caseKind": "reasoning",
   "route": "claude-cli",
   "modelRequested": "haiku",
   "modelReported": "claude-haiku-4-5-20251001",
   "effort": "default",
   "rep": 3,
   "startedAt": "2026-10-05T20:42:14.441Z",
   "finishedAt": "2026-10-05T20:42:18.039Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 2950,
   "totalMs": 3355,
   "modelTurnMs": 2514,
   "inputTokens": 3753,
   "outputTokens": 291,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 284,
   "passed": true,
   "checks": 1,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 4,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "shift-arithmetic",
   "caseTitle": "Multi-step shift arithmetic",
   "caseKind": "reasoning",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 3,
   "startedAt": "2026-10-05T20:42:18.040Z",
   "finishedAt": "2026-10-05T20:42:20.594Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 1080,
   "totalMs": 2301,
   "modelTurnMs": 1363,
   "inputTokens": 2022,
   "outputTokens": 90,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 557,
   "reasoningTokens": 0,
   "passed": false,
   "checks": 1,
   "failures": [
    "Output does not exactly match the expected text."
   ],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 146,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS."
   ],
   "failedOutputLastLine": "2292",
   "failedOutputLineCount": 5
  },
  {
   "batch": "batch-1",
   "caseId": "shift-arithmetic",
   "caseTitle": "Multi-step shift arithmetic",
   "caseKind": "reasoning",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "default",
   "rep": 3,
   "startedAt": "2026-10-05T20:42:20.595Z",
   "finishedAt": "2026-10-05T20:42:23.657Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 2393,
   "totalMs": 2830,
   "modelTurnMs": 1851,
   "inputTokens": 2018,
   "outputTokens": 60,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 553,
   "reasoningTokens": 56,
   "passed": true,
   "checks": 1,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 4,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "shift-arithmetic",
   "caseTitle": "Multi-step shift arithmetic",
   "caseKind": "reasoning",
   "route": "claude-cli",
   "modelRequested": "fable",
   "modelReported": "claude-fable-5-1",
   "effort": "default",
   "rep": 3,
   "startedAt": "2026-10-05T20:42:23.658Z",
   "finishedAt": "2026-10-05T20:42:25.306Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 1015,
   "totalMs": 1409,
   "modelTurnMs": 576,
   "inputTokens": 3170,
   "outputTokens": 4,
   "cacheReadTokens": 2968,
   "cacheWriteTokens": 200,
   "reasoningTokens": 0,
   "passed": true,
   "checks": 1,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 4,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "shift-arithmetic",
   "caseTitle": "Multi-step shift arithmetic",
   "caseKind": "reasoning",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "low",
   "rep": 3,
   "startedAt": "2026-10-05T20:42:25.306Z",
   "finishedAt": "2026-10-05T20:42:28.369Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 2499,
   "totalMs": 2825,
   "modelTurnMs": 1768,
   "inputTokens": 2018,
   "outputTokens": 60,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 553,
   "reasoningTokens": 56,
   "passed": true,
   "checks": 1,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 4,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "shift-arithmetic",
   "caseTitle": "Multi-step shift arithmetic",
   "caseKind": "reasoning",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "high",
   "rep": 3,
   "startedAt": "2026-10-05T20:42:28.370Z",
   "finishedAt": "2026-10-05T20:42:31.276Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 2295,
   "totalMs": 2678,
   "modelTurnMs": 1724,
   "inputTokens": 2018,
   "outputTokens": 60,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 553,
   "reasoningTokens": 56,
   "passed": true,
   "checks": 1,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 4,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "flatten-iterative",
   "caseTitle": "Refactor recursion to iteration",
   "caseKind": "refactor",
   "route": "claude-cli",
   "modelRequested": "haiku",
   "modelReported": "claude-haiku-4-5-20251001",
   "effort": "default",
   "rep": 3,
   "startedAt": "2026-10-05T20:42:31.277Z",
   "finishedAt": "2026-10-05T20:42:49.548Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 16317,
   "totalMs": 17750,
   "modelTurnMs": 17085,
   "inputTokens": 3789,
   "outputTokens": 2099,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 1893,
   "passed": true,
   "checks": 10,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 539,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "flatten-iterative",
   "caseTitle": "Refactor recursion to iteration",
   "caseKind": "refactor",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 3,
   "startedAt": "2026-10-05T20:42:49.549Z",
   "finishedAt": "2026-10-05T20:42:57.784Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 6392,
   "totalMs": 7725,
   "modelTurnMs": 7007,
   "inputTokens": 2081,
   "outputTokens": 745,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 616,
   "reasoningTokens": 542,
   "passed": true,
   "checks": 10,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 487,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "flatten-iterative",
   "caseTitle": "Refactor recursion to iteration",
   "caseKind": "refactor",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "default",
   "rep": 3,
   "startedAt": "2026-10-05T20:42:57.785Z",
   "finishedAt": "2026-10-05T20:43:07.092Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 7017,
   "totalMs": 8798,
   "modelTurnMs": 8077,
   "inputTokens": 2077,
   "outputTokens": 853,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 612,
   "reasoningTokens": 616,
   "passed": true,
   "checks": 10,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 574,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "flatten-iterative",
   "caseTitle": "Refactor recursion to iteration",
   "caseKind": "refactor",
   "route": "claude-cli",
   "modelRequested": "fable",
   "modelReported": "claude-fable-5-1",
   "effort": "default",
   "rep": 3,
   "startedAt": "2026-10-05T20:43:07.093Z",
   "finishedAt": "2026-10-05T20:43:17.434Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 7902,
   "totalMs": 9828,
   "modelTurnMs": 9052,
   "inputTokens": 3229,
   "outputTokens": 908,
   "cacheReadTokens": 3027,
   "cacheWriteTokens": 200,
   "reasoningTokens": 672,
   "passed": true,
   "checks": 10,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 572,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "flatten-iterative",
   "caseTitle": "Refactor recursion to iteration",
   "caseKind": "refactor",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "low",
   "rep": 3,
   "startedAt": "2026-10-05T20:43:17.434Z",
   "finishedAt": "2026-10-05T20:43:24.579Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 4899,
   "totalMs": 6617,
   "modelTurnMs": 5884,
   "inputTokens": 2076,
   "outputTokens": 637,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 611,
   "reasoningTokens": 405,
   "passed": true,
   "checks": 10,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 557,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "flatten-iterative",
   "caseTitle": "Refactor recursion to iteration",
   "caseKind": "refactor",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "high",
   "rep": 3,
   "startedAt": "2026-10-05T20:43:24.580Z",
   "finishedAt": "2026-10-05T20:43:35.176Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 8308,
   "totalMs": 10080,
   "modelTurnMs": 9277,
   "inputTokens": 2078,
   "outputTokens": 904,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 613,
   "reasoningTokens": 668,
   "passed": true,
   "checks": 10,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 572,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "ticket-triage",
   "caseTitle": "Classify six support tickets",
   "caseKind": "classification",
   "route": "claude-cli",
   "modelRequested": "haiku",
   "modelReported": "claude-haiku-4-5-20251001",
   "effort": "default",
   "rep": 3,
   "startedAt": "2026-10-05T20:43:35.177Z",
   "finishedAt": "2026-10-05T20:43:39.798Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 3745,
   "totalMs": 4397,
   "modelTurnMs": 3505,
   "inputTokens": 3828,
   "outputTokens": 316,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 248,
   "passed": true,
   "checks": 2,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 122,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "ticket-triage",
   "caseTitle": "Classify six support tickets",
   "caseKind": "classification",
   "route": "claude-cli",
   "modelRequested": "sonnet",
   "modelReported": "claude-sonnet-5-5",
   "effort": "default",
   "rep": 3,
   "startedAt": "2026-10-05T20:43:39.798Z",
   "finishedAt": "2026-10-05T20:43:45.024Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 4466,
   "totalMs": 5001,
   "modelTurnMs": 4148,
   "inputTokens": 2149,
   "outputTokens": 44,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 684,
   "reasoningTokens": 0,
   "passed": true,
   "checks": 2,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 85,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "ticket-triage",
   "caseTitle": "Classify six support tickets",
   "caseKind": "classification",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "default",
   "rep": 3,
   "startedAt": "2026-10-05T20:43:45.025Z",
   "finishedAt": "2026-10-05T20:43:47.765Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 1916,
   "totalMs": 2512,
   "modelTurnMs": 1714,
   "inputTokens": 2145,
   "outputTokens": 44,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 680,
   "reasoningTokens": 0,
   "passed": true,
   "checks": 2,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 85,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "ticket-triage",
   "caseTitle": "Classify six support tickets",
   "caseKind": "classification",
   "route": "claude-cli",
   "modelRequested": "fable",
   "modelReported": "claude-fable-5-1",
   "effort": "default",
   "rep": 3,
   "startedAt": "2026-10-05T20:43:47.766Z",
   "finishedAt": "2026-10-05T20:43:49.551Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 947,
   "totalMs": 1560,
   "modelTurnMs": 722,
   "inputTokens": 3296,
   "outputTokens": 44,
   "cacheReadTokens": 3095,
   "cacheWriteTokens": 199,
   "reasoningTokens": 0,
   "passed": true,
   "checks": 2,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 85,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "ticket-triage",
   "caseTitle": "Classify six support tickets",
   "caseKind": "classification",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "low",
   "rep": 3,
   "startedAt": "2026-10-05T20:43:49.552Z",
   "finishedAt": "2026-10-05T20:43:52.181Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 1761,
   "totalMs": 2403,
   "modelTurnMs": 1559,
   "inputTokens": 2146,
   "outputTokens": 44,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 681,
   "reasoningTokens": 0,
   "passed": true,
   "checks": 2,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 85,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS."
   ]
  },
  {
   "batch": "batch-1",
   "caseId": "ticket-triage",
   "caseTitle": "Classify six support tickets",
   "caseKind": "classification",
   "route": "claude-cli",
   "modelRequested": "opus",
   "modelReported": "claude-opus-5-5",
   "effort": "high",
   "rep": 3,
   "startedAt": "2026-10-05T20:43:52.182Z",
   "finishedAt": "2026-10-05T20:43:55.272Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 2234,
   "totalMs": 2861,
   "modelTurnMs": 2012,
   "inputTokens": 2145,
   "outputTokens": 78,
   "cacheReadTokens": 1463,
   "cacheWriteTokens": 680,
   "reasoningTokens": 34,
   "passed": true,
   "checks": 2,
   "failures": [],
   "cliVersion": "2.1.286 (Claude Code)",
   "speed": "standard",
   "outputChars": 85,
   "notes": [
    "Minimal context via --safe-mode with all setting sources empty. The CLI still adds its own default system prompt, so the context hash is unknown.",
    "Output cap enforced through CLAUDE_CODE_MAX_OUTPUT_TOKENS.",
    "Reasoning was observed but its content is never captured."
   ]
  },
  {
   "batch": "batch-2",
   "caseId": "fix-median",
   "caseTitle": "Fix a buggy median function",
   "caseKind": "code-fix",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "low",
   "rep": 1,
   "startedAt": "2026-10-05T20:36:30.184Z",
   "finishedAt": "2026-10-05T20:36:41.053Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 8497,
   "totalMs": 10468,
   "modelTurnMs": 9516,
   "inputTokens": 12094,
   "outputTokens": 218,
   "cacheReadTokens": 8960,
   "cacheWriteTokens": 0,
   "reasoningTokens": 103,
   "passed": true,
   "checks": 8,
   "failures": [],
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 345,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ]
  },
  {
   "batch": "batch-2",
   "caseId": "fix-median",
   "caseTitle": "Fix a buggy median function",
   "caseKind": "code-fix",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "medium",
   "rep": 1,
   "startedAt": "2026-10-05T20:36:41.053Z",
   "finishedAt": "2026-10-05T20:36:51.126Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 7590,
   "totalMs": 9564,
   "modelTurnMs": 8965,
   "inputTokens": 12094,
   "outputTokens": 211,
   "cacheReadTokens": 8960,
   "cacheWriteTokens": 0,
   "reasoningTokens": 96,
   "passed": true,
   "checks": 8,
   "failures": [],
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 314,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ]
  },
  {
   "batch": "batch-2",
   "caseId": "fix-median",
   "caseTitle": "Fix a buggy median function",
   "caseKind": "code-fix",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "high",
   "rep": 1,
   "startedAt": "2026-10-05T20:36:51.127Z",
   "finishedAt": "2026-10-05T20:37:05.718Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 11995,
   "totalMs": 14143,
   "modelTurnMs": 13530,
   "inputTokens": 12094,
   "outputTokens": 353,
   "cacheReadTokens": 8960,
   "cacheWriteTokens": 0,
   "reasoningTokens": 239,
   "passed": true,
   "checks": 8,
   "failures": [],
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 306,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ]
  },
  {
   "batch": "batch-2",
   "caseId": "invoice-json",
   "caseTitle": "Extract invoice fields to JSON",
   "caseKind": "extraction",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "low",
   "rep": 1,
   "startedAt": "2026-10-05T20:37:05.719Z",
   "finishedAt": "2026-10-05T20:37:10.858Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 4383,
   "totalMs": 5138,
   "modelTurnMs": 4304,
   "inputTokens": 12149,
   "outputTokens": 42,
   "cacheReadTokens": 8960,
   "cacheWriteTokens": 0,
   "reasoningTokens": 0,
   "passed": true,
   "checks": 2,
   "failures": [],
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 116,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ]
  },
  {
   "batch": "batch-2",
   "caseId": "invoice-json",
   "caseTitle": "Extract invoice fields to JSON",
   "caseKind": "extraction",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "medium",
   "rep": 1,
   "startedAt": "2026-10-05T20:37:10.859Z",
   "finishedAt": "2026-10-05T20:37:15.855Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 4259,
   "totalMs": 4995,
   "modelTurnMs": 4403,
   "inputTokens": 12151,
   "outputTokens": 42,
   "cacheReadTokens": 8960,
   "cacheWriteTokens": 0,
   "reasoningTokens": 0,
   "passed": true,
   "checks": 2,
   "failures": [],
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 116,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ]
  },
  {
   "batch": "batch-2",
   "caseId": "invoice-json",
   "caseTitle": "Extract invoice fields to JSON",
   "caseKind": "extraction",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "high",
   "rep": 1,
   "startedAt": "2026-10-05T20:37:15.856Z",
   "finishedAt": "2026-10-05T20:37:20.417Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 3778,
   "totalMs": 4561,
   "modelTurnMs": 3928,
   "inputTokens": 12149,
   "outputTokens": 42,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 0,
   "passed": true,
   "checks": 2,
   "failures": [],
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 116,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ]
  },
  {
   "batch": "batch-2",
   "caseId": "shift-arithmetic",
   "caseTitle": "Multi-step shift arithmetic",
   "caseKind": "reasoning",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "low",
   "rep": 1,
   "startedAt": "2026-10-05T20:37:20.418Z",
   "finishedAt": "2026-10-05T20:37:25.532Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 4986,
   "totalMs": 5113,
   "modelTurnMs": 4575,
   "inputTokens": 12083,
   "outputTokens": 28,
   "cacheReadTokens": 8960,
   "cacheWriteTokens": 0,
   "reasoningTokens": 20,
   "passed": true,
   "checks": 1,
   "failures": [],
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 4,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ]
  },
  {
   "batch": "batch-2",
   "caseId": "shift-arithmetic",
   "caseTitle": "Multi-step shift arithmetic",
   "caseKind": "reasoning",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "medium",
   "rep": 1,
   "startedAt": "2026-10-05T20:37:25.532Z",
   "finishedAt": "2026-10-05T20:37:29.765Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 4113,
   "totalMs": 4229,
   "modelTurnMs": 3349,
   "inputTokens": 12083,
   "outputTokens": 28,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 20,
   "passed": true,
   "checks": 1,
   "failures": [],
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 4,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ]
  },
  {
   "batch": "batch-2",
   "caseId": "shift-arithmetic",
   "caseTitle": "Multi-step shift arithmetic",
   "caseKind": "reasoning",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "high",
   "rep": 1,
   "startedAt": "2026-10-05T20:37:29.766Z",
   "finishedAt": "2026-10-05T20:37:35.372Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 5323,
   "totalMs": 5603,
   "modelTurnMs": 4923,
   "inputTokens": 12085,
   "outputTokens": 28,
   "cacheReadTokens": 8960,
   "cacheWriteTokens": 0,
   "reasoningTokens": 20,
   "passed": true,
   "checks": 1,
   "failures": [],
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 4,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ]
  },
  {
   "batch": "batch-2",
   "caseId": "flatten-iterative",
   "caseTitle": "Refactor recursion to iteration",
   "caseKind": "refactor",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "low",
   "rep": 1,
   "startedAt": "2026-10-05T20:37:35.373Z",
   "finishedAt": "2026-10-05T20:37:45.184Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 6782,
   "totalMs": 9413,
   "modelTurnMs": 8850,
   "inputTokens": 12124,
   "outputTokens": 214,
   "cacheReadTokens": 8960,
   "cacheWriteTokens": 0,
   "reasoningTokens": 54,
   "passed": true,
   "checks": 10,
   "failures": [],
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 601,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ]
  },
  {
   "batch": "batch-2",
   "caseId": "flatten-iterative",
   "caseTitle": "Refactor recursion to iteration",
   "caseKind": "refactor",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "medium",
   "rep": 1,
   "startedAt": "2026-10-05T20:37:45.185Z",
   "finishedAt": "2026-10-05T20:38:11.053Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 17816,
   "totalMs": 25461,
   "modelTurnMs": 24859,
   "inputTokens": 12124,
   "outputTokens": 261,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 101,
   "passed": true,
   "checks": 10,
   "failures": [],
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 601,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ]
  },
  {
   "batch": "batch-2",
   "caseId": "flatten-iterative",
   "caseTitle": "Refactor recursion to iteration",
   "caseKind": "refactor",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "high",
   "rep": 1,
   "startedAt": "2026-10-05T20:38:11.054Z",
   "finishedAt": "2026-10-05T20:38:28.067Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 13981,
   "totalMs": 16575,
   "modelTurnMs": 15195,
   "inputTokens": 12124,
   "outputTokens": 391,
   "cacheReadTokens": 8960,
   "cacheWriteTokens": 0,
   "reasoningTokens": 227,
   "passed": true,
   "checks": 10,
   "failures": [],
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 614,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ]
  },
  {
   "batch": "batch-2",
   "caseId": "ticket-triage",
   "caseTitle": "Classify six support tickets",
   "caseKind": "classification",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "low",
   "rep": 1,
   "startedAt": "2026-10-05T20:38:28.067Z",
   "finishedAt": "2026-10-05T20:38:32.912Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 4242,
   "totalMs": 4841,
   "modelTurnMs": 3854,
   "inputTokens": 12161,
   "outputTokens": 35,
   "cacheReadTokens": 8960,
   "cacheWriteTokens": 0,
   "reasoningTokens": 0,
   "passed": true,
   "checks": 2,
   "failures": [],
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 85,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ]
  },
  {
   "batch": "batch-2",
   "caseId": "ticket-triage",
   "caseTitle": "Classify six support tickets",
   "caseKind": "classification",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "medium",
   "rep": 1,
   "startedAt": "2026-10-05T20:38:32.912Z",
   "finishedAt": "2026-10-05T20:38:37.008Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 3363,
   "totalMs": 4095,
   "modelTurnMs": 3559,
   "inputTokens": 12159,
   "outputTokens": 35,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 0,
   "passed": true,
   "checks": 2,
   "failures": [],
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 85,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ]
  },
  {
   "batch": "batch-2",
   "caseId": "ticket-triage",
   "caseTitle": "Classify six support tickets",
   "caseKind": "classification",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "high",
   "rep": 1,
   "startedAt": "2026-10-05T20:38:37.009Z",
   "finishedAt": "2026-10-05T20:38:43.257Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 5598,
   "totalMs": 6247,
   "modelTurnMs": 5368,
   "inputTokens": 12159,
   "outputTokens": 35,
   "cacheReadTokens": 8960,
   "cacheWriteTokens": 0,
   "reasoningTokens": 0,
   "passed": true,
   "checks": 2,
   "failures": [],
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 85,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ]
  },
  {
   "batch": "batch-2",
   "caseId": "fix-median",
   "caseTitle": "Fix a buggy median function",
   "caseKind": "code-fix",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "low",
   "rep": 2,
   "startedAt": "2026-10-05T20:38:43.257Z",
   "finishedAt": "2026-10-05T20:38:53.121Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 7623,
   "totalMs": 9455,
   "modelTurnMs": 8890,
   "inputTokens": 12096,
   "outputTokens": 193,
   "cacheReadTokens": 8960,
   "cacheWriteTokens": 0,
   "reasoningTokens": 78,
   "passed": true,
   "checks": 8,
   "failures": [],
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 314,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ]
  },
  {
   "batch": "batch-2",
   "caseId": "fix-median",
   "caseTitle": "Fix a buggy median function",
   "caseKind": "code-fix",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "medium",
   "rep": 2,
   "startedAt": "2026-10-05T20:38:53.121Z",
   "finishedAt": "2026-10-05T20:39:05.139Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 9819,
   "totalMs": 11591,
   "modelTurnMs": 11021,
   "inputTokens": 12096,
   "outputTokens": 237,
   "cacheReadTokens": 11136,
   "cacheWriteTokens": 0,
   "reasoningTokens": 122,
   "passed": true,
   "checks": 8,
   "failures": [],
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 314,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ]
  },
  {
   "batch": "batch-2",
   "caseId": "fix-median",
   "caseTitle": "Fix a buggy median function",
   "caseKind": "code-fix",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "high",
   "rep": 2,
   "startedAt": "2026-10-05T20:39:05.140Z",
   "finishedAt": "2026-10-05T20:39:17.905Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 10445,
   "totalMs": 12348,
   "modelTurnMs": 11659,
   "inputTokens": 12094,
   "outputTokens": 331,
   "cacheReadTokens": 8960,
   "cacheWriteTokens": 0,
   "reasoningTokens": 216,
   "passed": true,
   "checks": 8,
   "failures": [],
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 314,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ]
  },
  {
   "batch": "batch-2",
   "caseId": "invoice-json",
   "caseTitle": "Extract invoice fields to JSON",
   "caseKind": "extraction",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "low",
   "rep": 2,
   "startedAt": "2026-10-05T20:39:17.905Z",
   "finishedAt": "2026-10-05T20:39:25.296Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 6450,
   "totalMs": 7390,
   "modelTurnMs": 6668,
   "inputTokens": 12149,
   "outputTokens": 42,
   "cacheReadTokens": 8960,
   "cacheWriteTokens": 0,
   "reasoningTokens": 0,
   "passed": true,
   "checks": 2,
   "failures": [],
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 116,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ]
  },
  {
   "batch": "batch-2",
   "caseId": "invoice-json",
   "caseTitle": "Extract invoice fields to JSON",
   "caseKind": "extraction",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "medium",
   "rep": 2,
   "startedAt": "2026-10-05T20:39:25.296Z",
   "finishedAt": "2026-10-05T20:39:30.946Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 4868,
   "totalMs": 5649,
   "modelTurnMs": 5071,
   "inputTokens": 12149,
   "outputTokens": 42,
   "cacheReadTokens": 8960,
   "cacheWriteTokens": 0,
   "reasoningTokens": 0,
   "passed": true,
   "checks": 2,
   "failures": [],
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 116,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ]
  },
  {
   "batch": "batch-2",
   "caseId": "invoice-json",
   "caseTitle": "Extract invoice fields to JSON",
   "caseKind": "extraction",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "high",
   "rep": 2,
   "startedAt": "2026-10-05T20:39:30.946Z",
   "finishedAt": "2026-10-05T20:39:35.747Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 3973,
   "totalMs": 4800,
   "modelTurnMs": 4011,
   "inputTokens": 12151,
   "outputTokens": 42,
   "cacheReadTokens": 8960,
   "cacheWriteTokens": 0,
   "reasoningTokens": 0,
   "passed": true,
   "checks": 2,
   "failures": [],
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 116,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ]
  },
  {
   "batch": "batch-2",
   "caseId": "shift-arithmetic",
   "caseTitle": "Multi-step shift arithmetic",
   "caseKind": "reasoning",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "low",
   "rep": 2,
   "startedAt": "2026-10-05T20:39:35.748Z",
   "finishedAt": "2026-10-05T20:39:40.397Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 4497,
   "totalMs": 4649,
   "modelTurnMs": 3972,
   "inputTokens": 12085,
   "outputTokens": 28,
   "cacheReadTokens": 8960,
   "cacheWriteTokens": 0,
   "reasoningTokens": 20,
   "passed": true,
   "checks": 1,
   "failures": [],
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 4,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ]
  },
  {
   "batch": "batch-2",
   "caseId": "shift-arithmetic",
   "caseTitle": "Multi-step shift arithmetic",
   "caseKind": "reasoning",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "medium",
   "rep": 2,
   "startedAt": "2026-10-05T20:39:40.398Z",
   "finishedAt": "2026-10-05T20:39:46.366Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 5774,
   "totalMs": 5968,
   "modelTurnMs": 5185,
   "inputTokens": 12083,
   "outputTokens": 28,
   "cacheReadTokens": 8960,
   "cacheWriteTokens": 0,
   "reasoningTokens": 20,
   "passed": true,
   "checks": 1,
   "failures": [],
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 4,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ]
  },
  {
   "batch": "batch-2",
   "caseId": "shift-arithmetic",
   "caseTitle": "Multi-step shift arithmetic",
   "caseKind": "reasoning",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "high",
   "rep": 2,
   "startedAt": "2026-10-05T20:39:46.367Z",
   "finishedAt": "2026-10-05T20:39:50.419Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 3797,
   "totalMs": 4052,
   "modelTurnMs": 3478,
   "inputTokens": 12085,
   "outputTokens": 33,
   "cacheReadTokens": 8960,
   "cacheWriteTokens": 0,
   "reasoningTokens": 25,
   "passed": true,
   "checks": 1,
   "failures": [],
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 4,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ]
  },
  {
   "batch": "batch-2",
   "caseId": "flatten-iterative",
   "caseTitle": "Refactor recursion to iteration",
   "caseKind": "refactor",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "low",
   "rep": 2,
   "startedAt": "2026-10-05T20:39:50.419Z",
   "finishedAt": "2026-10-05T20:39:58.707Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 5296,
   "totalMs": 7986,
   "modelTurnMs": 7445,
   "inputTokens": 12122,
   "outputTokens": 225,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 57,
   "passed": true,
   "checks": 10,
   "failures": [],
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 636,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ]
  },
  {
   "batch": "batch-2",
   "caseId": "flatten-iterative",
   "caseTitle": "Refactor recursion to iteration",
   "caseKind": "refactor",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "medium",
   "rep": 2,
   "startedAt": "2026-10-05T20:39:58.707Z",
   "finishedAt": "2026-10-05T20:40:12.729Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 10358,
   "totalMs": 13700,
   "modelTurnMs": 12790,
   "inputTokens": 12126,
   "outputTokens": 295,
   "cacheReadTokens": 8960,
   "cacheWriteTokens": 0,
   "reasoningTokens": 137,
   "passed": true,
   "checks": 10,
   "failures": [],
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 584,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ]
  },
  {
   "batch": "batch-2",
   "caseId": "flatten-iterative",
   "caseTitle": "Refactor recursion to iteration",
   "caseKind": "refactor",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "high",
   "rep": 2,
   "startedAt": "2026-10-05T20:40:12.730Z",
   "finishedAt": "2026-10-05T20:40:27.129Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 11657,
   "totalMs": 14083,
   "modelTurnMs": 13462,
   "inputTokens": 12124,
   "outputTokens": 351,
   "cacheReadTokens": 11136,
   "cacheWriteTokens": 0,
   "reasoningTokens": 191,
   "passed": true,
   "checks": 10,
   "failures": [],
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 601,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ]
  },
  {
   "batch": "batch-2",
   "caseId": "ticket-triage",
   "caseTitle": "Classify six support tickets",
   "caseKind": "classification",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "low",
   "rep": 2,
   "startedAt": "2026-10-05T20:40:27.130Z",
   "finishedAt": "2026-10-05T20:40:31.786Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 4022,
   "totalMs": 4655,
   "modelTurnMs": 4131,
   "inputTokens": 12163,
   "outputTokens": 35,
   "cacheReadTokens": 8960,
   "cacheWriteTokens": 0,
   "reasoningTokens": 0,
   "passed": true,
   "checks": 2,
   "failures": [],
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 85,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ]
  },
  {
   "batch": "batch-2",
   "caseId": "ticket-triage",
   "caseTitle": "Classify six support tickets",
   "caseKind": "classification",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "medium",
   "rep": 2,
   "startedAt": "2026-10-05T20:40:31.786Z",
   "finishedAt": "2026-10-05T20:40:36.099Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 3670,
   "totalMs": 4313,
   "modelTurnMs": 3815,
   "inputTokens": 12161,
   "outputTokens": 35,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 0,
   "passed": true,
   "checks": 2,
   "failures": [],
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 85,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ]
  },
  {
   "batch": "batch-2",
   "caseId": "ticket-triage",
   "caseTitle": "Classify six support tickets",
   "caseKind": "classification",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "high",
   "rep": 2,
   "startedAt": "2026-10-05T20:40:36.100Z",
   "finishedAt": "2026-10-05T20:40:40.350Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 3636,
   "totalMs": 4249,
   "modelTurnMs": 3749,
   "inputTokens": 12159,
   "outputTokens": 35,
   "cacheReadTokens": 8960,
   "cacheWriteTokens": 0,
   "reasoningTokens": 0,
   "passed": true,
   "checks": 2,
   "failures": [],
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 85,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ]
  },
  {
   "batch": "batch-3",
   "caseId": "fix-median",
   "caseTitle": "Fix a buggy median function",
   "caseKind": "code-fix",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "medium",
   "rep": 3,
   "startedAt": "2026-10-06T13:15:38.084Z",
   "finishedAt": "2026-10-06T13:15:51.919Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 9200,
   "totalMs": 13528,
   "modelTurnMs": 12887,
   "inputTokens": 12094,
   "outputTokens": 251,
   "cacheReadTokens": 8960,
   "cacheWriteTokens": 0,
   "reasoningTokens": 136,
   "passed": true,
   "checks": 8,
   "failures": [],
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 314,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ]
  },
  {
   "batch": "batch-3",
   "caseId": "fix-median",
   "caseTitle": "Fix a buggy median function",
   "caseKind": "code-fix",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "high",
   "rep": 3,
   "startedAt": "2026-10-06T13:15:51.919Z",
   "finishedAt": "2026-10-06T13:16:11.154Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 16070,
   "totalMs": 18695,
   "modelTurnMs": 18214,
   "inputTokens": 12094,
   "outputTokens": 393,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 282,
   "passed": true,
   "checks": 8,
   "failures": [],
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 298,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ]
  },
  {
   "batch": "batch-3",
   "caseId": "invoice-json",
   "caseTitle": "Extract invoice fields to JSON",
   "caseKind": "extraction",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "medium",
   "rep": 3,
   "startedAt": "2026-10-06T13:16:11.154Z",
   "finishedAt": "2026-10-06T13:16:16.299Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 4383,
   "totalMs": 5145,
   "modelTurnMs": 4340,
   "inputTokens": 12147,
   "outputTokens": 42,
   "cacheReadTokens": 3840,
   "cacheWriteTokens": 0,
   "reasoningTokens": 0,
   "passed": true,
   "checks": 2,
   "failures": [],
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 116,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ]
  },
  {
   "batch": "batch-3",
   "caseId": "invoice-json",
   "caseTitle": "Extract invoice fields to JSON",
   "caseKind": "extraction",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "high",
   "rep": 3,
   "startedAt": "2026-10-06T13:16:16.300Z",
   "finishedAt": "2026-10-06T13:16:21.333Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 4202,
   "totalMs": 5033,
   "modelTurnMs": 4316,
   "inputTokens": 12149,
   "outputTokens": 42,
   "cacheReadTokens": 8960,
   "cacheWriteTokens": 0,
   "reasoningTokens": 0,
   "passed": true,
   "checks": 2,
   "failures": [],
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 116,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ]
  },
  {
   "batch": "batch-3",
   "caseId": "shift-arithmetic",
   "caseTitle": "Multi-step shift arithmetic",
   "caseKind": "reasoning",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "medium",
   "rep": 3,
   "startedAt": "2026-10-06T13:16:21.333Z",
   "finishedAt": "2026-10-06T13:16:26.491Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 5047,
   "totalMs": 5157,
   "modelTurnMs": 4664,
   "inputTokens": 12087,
   "outputTokens": 27,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 19,
   "passed": true,
   "checks": 1,
   "failures": [],
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 4,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ]
  },
  {
   "batch": "batch-3",
   "caseId": "shift-arithmetic",
   "caseTitle": "Multi-step shift arithmetic",
   "caseKind": "reasoning",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "high",
   "rep": 3,
   "startedAt": "2026-10-06T13:16:26.492Z",
   "finishedAt": "2026-10-06T13:16:30.644Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 4082,
   "totalMs": 4152,
   "modelTurnMs": 3670,
   "inputTokens": 12083,
   "outputTokens": 29,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 21,
   "passed": true,
   "checks": 1,
   "failures": [],
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 4,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ]
  },
  {
   "batch": "batch-3",
   "caseId": "flatten-iterative",
   "caseTitle": "Refactor recursion to iteration",
   "caseKind": "refactor",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "medium",
   "rep": 3,
   "startedAt": "2026-10-06T13:16:30.645Z",
   "finishedAt": "2026-10-06T13:16:44.908Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 11005,
   "totalMs": 13930,
   "modelTurnMs": 13415,
   "inputTokens": 12124,
   "outputTokens": 285,
   "cacheReadTokens": 8960,
   "cacheWriteTokens": 0,
   "reasoningTokens": 117,
   "passed": true,
   "checks": 10,
   "failures": [],
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 636,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ]
  },
  {
   "batch": "batch-3",
   "caseId": "flatten-iterative",
   "caseTitle": "Refactor recursion to iteration",
   "caseKind": "refactor",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "high",
   "rep": 3,
   "startedAt": "2026-10-06T13:16:44.909Z",
   "finishedAt": "2026-10-06T13:17:04.683Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 16372,
   "totalMs": 19515,
   "modelTurnMs": 18931,
   "inputTokens": 12122,
   "outputTokens": 457,
   "cacheReadTokens": 8960,
   "cacheWriteTokens": 0,
   "reasoningTokens": 289,
   "passed": true,
   "checks": 10,
   "failures": [],
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 636,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ]
  },
  {
   "batch": "batch-3",
   "caseId": "ticket-triage",
   "caseTitle": "Classify six support tickets",
   "caseKind": "classification",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "medium",
   "rep": 3,
   "startedAt": "2026-10-06T13:17:04.684Z",
   "finishedAt": "2026-10-06T13:17:08.789Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 3471,
   "totalMs": 4105,
   "modelTurnMs": 3642,
   "inputTokens": 12163,
   "outputTokens": 35,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 0,
   "passed": true,
   "checks": 2,
   "failures": [],
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 85,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ]
  },
  {
   "batch": "batch-3",
   "caseId": "ticket-triage",
   "caseTitle": "Classify six support tickets",
   "caseKind": "classification",
   "route": "codex-cli",
   "modelRequested": "gpt-6.1-sol",
   "modelReported": "gpt-6.1-sol",
   "effort": "high",
   "rep": 3,
   "startedAt": "2026-10-06T13:17:08.789Z",
   "finishedAt": "2026-10-06T13:17:14.069Z",
   "status": "completed",
   "error": null,
   "firstUsefulMs": 4666,
   "totalMs": 5279,
   "modelTurnMs": 4679,
   "inputTokens": 12159,
   "outputTokens": 35,
   "cacheReadTokens": 0,
   "cacheWriteTokens": 0,
   "reasoningTokens": 0,
   "passed": true,
   "checks": 2,
   "failures": [],
   "cliVersion": "codex-cli 0.160.0",
   "speed": null,
   "outputChars": 85,
   "notes": [
    "Provider prompt caching follows Codex defaults; observed cache counters are recorded and cold caching is not claimed.",
    "Codex adds its own default system prompt and tool context even with minimal settings; the context hash is unknown.",
    "Reasoning text is never captured; only visible answer text and token counts.",
    "Codex built-in tool schemas may remain in the model request; this runner denies tool actions and records observed tool calls.",
    "Codex CLI uses its provider output limit; a hard output token cap is not exposed by this installed app-server."
   ]
  }
 ],
 "protocolNote": "The run protocol was declared before the first run. Its public summary is the Method section of the study page."
}
